{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,4,30]],"date-time":"2025-04-30T04:22:59Z","timestamp":1745986979670,"version":"3.40.4"},"reference-count":77,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2013,2,14]],"date-time":"2013-02-14T00:00:00Z","timestamp":1360800000000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2013,12]]},"DOI":"10.1007\/s10772-013-9190-8","type":"journal-article","created":{"date-parts":[[2013,2,13]],"date-time":"2013-02-13T12:44:45Z","timestamp":1360759485000},"page":"381-401","source":"Crossref","is-referenced-by-count":0,"title":["A unified framework for domain independent online speaker indexing in eigen-voice space using an index tree of reference models"],"prefix":"10.1007","volume":"16","author":[{"given":"M. H.","family":"Moattar","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"M. M.","family":"Homayounpour","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2013,2,14]]},"reference":[{"issue":"8","key":"9190_CR1","doi-asserted-by":"crossref","first-page":"649","DOI":"10.1109\/LSP.2004.831666","volume":"11","author":"J. Ajmera","year":"2004","unstructured":"Ajmera, J., McCowan, I., & Bourlard, H. (2004). Robust speaker change detection. IEEE Signal Processing Letters, 11(8), 649\u2013651.","journal-title":"IEEE Signal Processing Letters"},{"key":"9190_CR2","volume-title":"III jornadas en tecnologia del habla","author":"X. Anguera","year":"2004","unstructured":"Anguera, X., & Hernando, J. (2004). XBIC: nueva medida para segmentacion de locutor hacia el indexado automatico de la senal de voz. In III jornadas en tecnologia del habla, Valencia, Spain."},{"key":"9190_CR3","volume-title":"Second international workshop on multimodal user authentication","author":"X. Anguera","year":"2006","unstructured":"Anguera, X., Wooters, C., & Hernando, J. (2006). Frame purification for cluster comparison in speaker diarization. In Second international workshop on multimodal user authentication."},{"key":"9190_CR4","first-page":"21","volume-title":"15th conf. uncertainty artif. intell.","author":"H. Attias","year":"1999","unstructured":"Attias, H. (1999). Inferring parameters and structure of latent variable models by variational Bayes. In 15th conf. uncertainty artif. intell., Stockholm, Sweden (pp. 21\u201330)."},{"issue":"5","key":"9190_CR5","doi-asserted-by":"crossref","first-page":"1505","DOI":"10.1109\/TASL.2006.878261","volume":"14","author":"C. Barras","year":"2006","unstructured":"Barras, C., Zhu, X., Meignier, S., & Gauvain, J. L. (2006). Multistage speaker diarization of broadcast news. IEEE Transactions on Audio, Speech, and Language Processing, 14(5), 1505\u20131512.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"9190_CR6","doi-asserted-by":"crossref","first-page":"70","DOI":"10.1145\/951676.951690","volume-title":"ACM workshop on multimedia databases","author":"S. Berrani","year":"2003","unstructured":"Berrani, S., Amsaleg, L., & Gros, P. (2003). Robust content-based image searches for copyright protection. In ACM workshop on multimedia databases, New Orleans, USA (pp. 70\u201377)."},{"key":"9190_CR7","unstructured":"Bijankhan, M. (2002). Great farsdat database (Technical report). Iran Research center on Intelligent Signal Processing."},{"issue":"1\u20132","key":"9190_CR8","doi-asserted-by":"crossref","first-page":"177","DOI":"10.1016\/0167-6393(95)00013-E","volume":"17","author":"F. Bimbot","year":"1995","unstructured":"Bimbot, F., Magrin-Chagnolleau, I., & Mathan, L. (1995). Second order statistical measures for text-independent speaker identification. Speech Communication, 17(1\u20132), 177\u2013192.","journal-title":"Speech Communication"},{"key":"9190_CR9","first-page":"4081","volume-title":"ICASSP","author":"C. Boehm","year":"2009","unstructured":"Boehm, C., & Pernkopf, F. (2009). Effective metric-based speaker segmentation in the frequency domain. In ICASSP (pp. 4081\u20134084)."},{"key":"9190_CR10","first-page":"645","volume-title":"Proc. of ICASSP","author":"S. S. Chen","year":"1998","unstructured":"Chen, S. S., & Gopalakrishnan, P. S. (1998). Clustering via the Bayesian information criterion with applications in speech recognition. In Proc. of ICASSP, USA (Vol.\u00a02, pp. 645\u2013648)."},{"key":"9190_CR11","first-page":"742","volume-title":"Interspeech","author":"K. Chen","year":"2000","unstructured":"Chen, K., et al. (2000). Fast speaker adaptation using eigenspace-based maximum likelihood linear regression. In Interspeech (pp. 742\u2013745)."},{"key":"9190_CR12","first-page":"4089","volume-title":"ICASSP","author":"S. M. Chu","year":"2009","unstructured":"Chu, S. M., Tang, H., & Huang, T. S. (2009). Fishervoice and semi-supervised speaker clustering. In ICASSP (pp. 4089\u20134092)."},{"key":"9190_CR13","first-page":"33","volume-title":"Proc. ICASSP","author":"M. Davy","year":"2000","unstructured":"Davy, M., Doncarli, C., & Tourneret, J. (2000). Supervised classification using MCMC methods. In Proc. ICASSP (pp. 33\u201336)."},{"issue":"1\u20132","key":"9190_CR14","doi-asserted-by":"crossref","first-page":"111","DOI":"10.1016\/S0167-6393(00)00027-3","volume":"32","author":"P. Delacourt","year":"2000","unstructured":"Delacourt, P., & Wellekens, C. J. (2000). DISTBIC: a speaker based segmentation for audio indexing. Speech Communication, 32(1\u20132), 111\u2013127.","journal-title":"Speech Communication"},{"issue":"1","key":"9190_CR15","first-page":"1C38","volume":"39","author":"A. Dempster","year":"1977","unstructured":"Dempster, A., Laird, N., & Rubin, D. (1977). Maximum likelihood from incomplete data via the EM algorithm. Journal of the Royal Statistical Society. Series B. Methodological, 39(1), 1C38.","journal-title":"Journal of the Royal Statistical Society. Series B. Methodological"},{"key":"9190_CR16","first-page":"872","volume-title":"ICASSP","author":"F. Desobry","year":"2003","unstructured":"Desobry, F., & Davy, M. (2003). Support vector-based online detection of abrupt changes. In ICASSP (Vol.\u00a05, pp. 872\u2013875)."},{"key":"9190_CR17","first-page":"4061","volume-title":"ICASSP","author":"N. W. D. Evans","year":"2009","unstructured":"Evans, N. W. D., Fredouille, C., & Bonastre, J. F. (2009). Speaker diarization using unsupervised discriminant analysis of inter-channel delay features. In ICASSP (pp. 4061\u20134064)."},{"key":"9190_CR18","first-page":"843","volume-title":"Proc. of interspeech","author":"D. Fernandez","year":"2009","unstructured":"Fernandez, D., Otero, P. L., & Mateo, C. G. (2009). An adaptive threshold computation for unsupervised speaker segmentation. In Proc. of interspeech, Brighton, UK (pp. 843\u2013849)."},{"key":"9190_CR19","unstructured":"Garofolo, J. S., Lamel, L. F., Fisher, W. M., Fiscus, J. G., Pallett, D.\u00a0S., & Dahlgren, N. L. (1993). In The DARPA TIMIT acoustic-phonetic continuous speech corpus CDROM. Linguistic data consortium."},{"key":"9190_CR20","unstructured":"Garofolo, J., et al. (2002). In NIST rich transcription 2002 evaluation: a preview. LREC."},{"key":"9190_CR21","first-page":"1335","volume-title":"Proc. of interspeech","author":"J. L. Gauvain","year":"1998","unstructured":"Gauvain, J. L., Lamel, L., & Adda, G. (1998). Partitioning and transcription of broadcast news data. In Proc. of interspeech, Sydney, Australia (Vol.\u00a04, pp. 1335\u20131338)."},{"key":"9190_CR22","volume-title":"Proc. of interspeech","author":"K. J. Han","year":"2007","unstructured":"Han, K. J., & Narayanan, S. (2007). A robust stopping criterion for agglomerative hierarchical clustering in a speaker diarization system. In Proc. of interspeech, Antwerp, Belgium."},{"key":"9190_CR23","doi-asserted-by":"crossref","first-page":"20","DOI":"10.21437\/Interspeech.2008-3","volume-title":"Interspeech","author":"K. J. Han","year":"2008","unstructured":"Han, K. J., & Narayanan, S. S. (2008). Agglomerative hierarchical speaker clustering using incremental Gaussian mixture cluster modeling. In Interspeech (pp. 20\u201323)."},{"key":"9190_CR24","volume-title":"International symposium on Chinese spoken language processing (ISCSLP)","author":"C. H. Huang","year":"2004","unstructured":"Huang, C. H., Chien, J. T., & Wang, H. M. (2004). A new eigenvoice approach to speaker adaptation. In International symposium on Chinese spoken language processing (ISCSLP), Hong Kong."},{"key":"9190_CR25","volume-title":"Proc. of interspeech","author":"J. Hung","year":"2000","unstructured":"Hung, J., Wang, H., & Lee, L. (2000). Automatic metric based speech segmentation for broadcast news via principal component analysis. In Proc. of interspeech, Beijing, China."},{"key":"9190_CR26","first-page":"4986","volume-title":"ICASSP","author":"K. Iso","year":"2010","unstructured":"Iso, K. (2010). Speaker clustering using vector quantization and spectral clustering. In ICASSP (pp. 4986\u20134989)."},{"key":"9190_CR27","volume-title":"Proc. of the international symposium on music information retrieval (ISMIR2000)","author":"O. Izmirli","year":"2000","unstructured":"Izmirli, O. (2000). Using a spectral flatness based feature for audio segmentation and retrieval (Abstract). In Proc. of the international symposium on music information retrieval (ISMIR2000), Plymouth, Massachusetts, USA."},{"key":"9190_CR28","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-4757-1904-8","volume-title":"Principal component analysis","author":"I. T. Jolliffe","year":"1986","unstructured":"Jolliffe, I. T. (1986). Principal component analysis. Berlin: Springer."},{"key":"9190_CR29","first-page":"1423","volume-title":"Proc. of ICASSP","author":"T. Kemp","year":"2000","unstructured":"Kemp, T., Schmidt, M., Westphal, M., & Waibel, A. (2000). Strategies for automatic segmentation of audio data. In Proc. of ICASSP, Istanbul, Turkey (Vol.\u00a03, pp. 1423\u20131426)."},{"key":"9190_CR30","first-page":"745","volume-title":"Proc. of ICASSP","author":"H. Kim","year":"2005","unstructured":"Kim, H., Elter, D., & Sikora, T. (2005). Hybrid speaker-based segmentation system using model-level clustering. In Proc. of ICASSP, Philadelphia, USA (Vol.\u00a0I, pp. 745\u2013748)."},{"key":"9190_CR31","first-page":"4093","volume-title":"ICASSP","author":"T. Koshinaka","year":"2009","unstructured":"Koshinaka, T., Nagatomo, K., & Shinoda, K. (2009). Online speaker clustering using incremental learning of an ergodic hidden Markov model. In ICASSP (pp. 4093\u20134096)."},{"issue":"5","key":"9190_CR32","doi-asserted-by":"crossref","first-page":"1091","DOI":"10.1016\/j.sigpro.2007.11.017","volume":"88","author":"M. Kotti","year":"2008","unstructured":"Kotti, M., Moschou, V., & Kotropoulos, C. (2008). Speaker segmentation and clustering. Signal Processing, 88(5), 1091\u20131124.","journal-title":"Signal Processing"},{"issue":"4","key":"9190_CR33","doi-asserted-by":"crossref","first-page":"695","DOI":"10.1109\/89.876308","volume":"8","author":"R. Kuhn","year":"2000","unstructured":"Kuhn, R., Junqua, J. C., Nguyen, P., & Niedzielski, N. (2000). Rapid speaker adaptation in eigenvoice space. IEEE Transactions on Speech and Audio Processing, 8(4), 695\u2013707.","journal-title":"IEEE Transactions on Speech and Audio Processing"},{"issue":"1","key":"9190_CR34","doi-asserted-by":"crossref","first-page":"79","DOI":"10.1214\/aoms\/1177729694","volume":"22","author":"S. Kullback","year":"1951","unstructured":"Kullback, S., & Leibler, R. A. (1951). On information and sufficiency. The Annals of Mathematical Statistics, 22(1), 79\u201386.","journal-title":"The Annals of Mathematical Statistics"},{"key":"9190_CR35","volume-title":"NIPS 16","author":"J. T. Kwok","year":"2004","unstructured":"Kwok, J. T., Mak, B., & Ho, S. (2004). Eigenvoice speaker adaptation via composite kernel PCA. In NIPS 16, Cambridge: MIT Press."},{"key":"9190_CR36","doi-asserted-by":"crossref","first-page":"1004","DOI":"10.1109\/TSA.2005.851981","volume":"13","author":"S. Kwon","year":"2004","unstructured":"Kwon, S., & Narayanan, S. (2004a). Unsupervised speaker indexing using generic models. IEEE Transactions on Speech and Audio Processing, 13, 1004\u20131013.","journal-title":"IEEE Transactions on Speech and Audio Processing"},{"key":"9190_CR37","doi-asserted-by":"crossref","first-page":"1517","DOI":"10.21437\/Interspeech.2004-571","volume-title":"Interspeech","author":"S. Kwon","year":"2004","unstructured":"Kwon, S., & Narayanan, S. (2004b). Speaker model quantization for unsupervised speaker indexing. In Interspeech (pp. 1517\u20131520)."},{"key":"9190_CR38","volume-title":"Proc. of interspeech","author":"J. F. Lopez","year":"2000","unstructured":"Lopez, J. F., & Ellis, D. P. W. (2000). Using acoustic condition clustering to improve acoustic change detection on broadcast news. In Proc. of interspeech, Beijing, China."},{"key":"9190_CR39","first-page":"602","volume-title":"Proc. of the ACM multimedia","author":"L. Lu","year":"2002","unstructured":"Lu, L., & Zhang, H. (2002). Speaker change detection and tracking in real-time news broadcast analysis. In Proc. of the ACM multimedia, France (pp. 602\u2013610)."},{"issue":"4","key":"9190_CR40","doi-asserted-by":"crossref","first-page":"332","DOI":"10.1007\/s00530-004-0160-5","volume":"10","author":"L. Lu","year":"2005","unstructured":"Lu, L., & Zhang, H. (2005). Unsupervised speaker segmentation and tracking in real-time audio content analysis. Multimedia Systems, 10(4), 332\u2013343.","journal-title":"Multimedia Systems"},{"key":"9190_CR41","first-page":"1333","volume-title":"Proc. ICSLP","author":"Y. Mami","year":"2002","unstructured":"Mami, Y., & Charlet, D. (2002). Speaker identification by location in an optimal space of anchor models. In Proc. ICSLP, Denver, Colorado, USA (pp. 1333\u20131336)."},{"key":"9190_CR42","volume-title":"Proc. of interspeech","author":"K. Markov","year":"2007","unstructured":"Markov, K., & Nakamura, S. (2007). Never-ending learning with dynamic hidden Markov network. In Proc. of interspeech."},{"key":"9190_CR43","doi-asserted-by":"crossref","first-page":"363","DOI":"10.21437\/Interspeech.2008-149","volume-title":"Interspeech","author":"K. Markov","year":"2008","unstructured":"Markov, K., & Nakamura, S. (2008). Improved novelty detection for online GMM based speaker diarization. In Interspeech, Brisbane, Australia (pp. 363\u2013366)."},{"key":"9190_CR44","first-page":"2549","volume-title":"17th European signal processing conference (Eusipco)","author":"M. H. Moattar","year":"2009","unstructured":"Moattar, M. H., & Homayounpour, M. M. (2009). A simple but efficient real-time voice activity detection algorithm. In 17th European signal processing conference (Eusipco) (pp. 2549\u20132553)."},{"key":"9190_CR45","first-page":"85","volume-title":"Proc. of ICASSP","author":"Y. Moh","year":"2003","unstructured":"Moh, Y., Nguyen, P., & Junqua, J. C. (2003). Toward domain independent clustering. In Proc. of ICASSP (Vol.\u00a0II, pp. 85\u201388)."},{"key":"9190_CR46","first-page":"895","volume-title":"Interspeech","author":"Y. K. Muthusamy","year":"1992","unstructured":"Muthusamy, Y. K., et al. (1992). The OGI multi-language telephone speech corpus. In Interspeech (pp. 895\u2013898)."},{"key":"9190_CR47","doi-asserted-by":"crossref","first-page":"355","DOI":"10.1007\/978-94-011-5014-9_12","volume-title":"Learning in graphical models","author":"R. M. Neal","year":"1998","unstructured":"Neal, R. M., & Hinton, G. E. (1998). A view of the EM algorithm that justifies incremental, sparse, and other variants. In Learning in graphical models (pp. 355\u2013368). Cambridge: MIT Press."},{"key":"9190_CR48","doi-asserted-by":"crossref","first-page":"36","DOI":"10.21437\/Interspeech.2008-7","volume-title":"Interspeech","author":"T. H. Nguyen","year":"2008","unstructured":"Nguyen, T. H., Cheng, E. S., & Li, H. (2008). T-test distance and clustering criterion for speaker diarization. In Interspeech (pp. 36\u201339)."},{"key":"9190_CR49","first-page":"4085","volume-title":"ICASSP","author":"T. H. Nguyen","year":"2009","unstructured":"Nguyen, T. H., Li, H., & Cheng, E. S. (2009). Cluster criterion functions in spectral subspace and their application in speaker clustering. In ICASSP (pp. 4085\u20134088)."},{"key":"9190_CR50","first-page":"2178","volume-title":"Interspeech","author":"H. Ning","year":"2006","unstructured":"Ning, H., Liu, M., Tang, H., & Huang, T. (2006). A spectral clustering approach to speaker diarization. In Interspeech (pp. 2178\u20132181)."},{"key":"9190_CR51","first-page":"172","volume-title":"ICASSP","author":"M. Nishida","year":"2003","unstructured":"Nishida, M., & Kawahara, T. (2003). Unsupervised speaker indexing using speaker model selection based on Bayesian information criterion. In ICASSP (Vol.\u00a01, pp. 172\u2013175)."},{"key":"9190_CR52","volume-title":"ICASSP","author":"M. Omar","year":"2005","unstructured":"Omar, M., Chaudhari, U., & Ramaswamy, G. (2005). Blind change detection for audio segmentation. In ICASSP."},{"key":"9190_CR53","first-page":"4970","volume-title":"Proc. of ICASSP","author":"P. L. Otero","year":"2010","unstructured":"Otero, P. L., Fernandez, L. D., & Mateo, C. G. (2010). Novel strategies for reducing the false alarm rate in a speaker segmentation system. In Proc. of ICASSP (pp. 4970\u20134973)."},{"key":"9190_CR54","volume-title":"Fundamentals of speech recognition","author":"L. R. Rabiner","year":"1993","unstructured":"Rabiner, L. R., & Juang, B. H. (1993). Fundamentals of speech recognition. Englewood Cliffs: Prentice-Hall."},{"issue":"1\u20132","key":"9190_CR55","doi-asserted-by":"crossref","first-page":"91","DOI":"10.1016\/0167-6393(95)00009-D","volume":"17","author":"D. A. Reynolds","year":"1995","unstructured":"Reynolds, D. A. (1995). Speaker identification and verification using Gaussian mixture speaker models. Speech Communication, 17(1\u20132), 91\u2013108.","journal-title":"Speech Communication"},{"issue":"1\u20133","key":"9190_CR56","doi-asserted-by":"crossref","first-page":"19","DOI":"10.1006\/dspr.1999.0361","volume":"10","author":"D. A. Reynolds","year":"2000","unstructured":"Reynolds, D. A., Quatieri, T. F., & Dunn, R. B. (2000). Speaker verification using adapted Gaussian mixture models. Digital Signal Processing, 10(1\u20133), 19\u201341.","journal-title":"Digital Signal Processing"},{"key":"9190_CR57","first-page":"48","volume-title":"IbPRIA, part II","author":"L. J. Rodriguez","year":"2007","unstructured":"Rodriguez, L. J., Penagarikano, M., & Bordel, G. (2007). A simple but effective approach to speaker tracking in broadcast news. In IbPRIA, part II (pp. 48\u201355)."},{"key":"9190_CR58","unstructured":"RT (2009). The 2009 (RT09) rich transcription meeting recognition evaluation plan. http:\/\/www.itl.nist.gov\/iad\/mig\/\/tests\/rt\/2009\/docs\/rt09-meeting-eval-plan-v2.pdf ."},{"key":"9190_CR59","doi-asserted-by":"crossref","first-page":"461","DOI":"10.1214\/aos\/1176344136","volume":"6","author":"G. Schwarz","year":"1978","unstructured":"Schwarz, G. (1978). Estimating the dimension of a model. The Annals of Statistics, 6, 461\u2013464.","journal-title":"The Annals of Statistics"},{"key":"9190_CR60","first-page":"97","volume-title":"DARPA speech recognition workshop","author":"M. A. Siegler","year":"1997","unstructured":"Siegler, M. A., Jain, U., Raj, B., & Stern, R. M. (1997). Automatic segmentation, classification and clustering of broadcast news audio. In DARPA speech recognition workshop, Chantilly (pp. 97\u201399)."},{"key":"9190_CR61","volume-title":"Eurospeech","author":"P. Sivakumaran","year":"2001","unstructured":"Sivakumaran, P., Fortuna, J., & Ariyaeeinia, A. (2001). On the use of the Bayesian information criterion in multiple speaker detection. In Eurospeech, Scandinavia."},{"key":"9190_CR62","first-page":"4982","volume-title":"ICASSP","author":"H. Sun","year":"2010","unstructured":"Sun, H., et al. (2010). Speaker diarization system for RT-07 and RT-09 meeting room audio. In ICASSP (pp. 4982\u20134985)."},{"key":"9190_CR63","first-page":"4101","volume-title":"ICASSP","author":"H. Tang","year":"2009","unstructured":"Tang, H., Chu, S. M., & Huang, T. S. (2009). Generative model-based speaker clustering via mixture of von Mises-Fisher distributions. In ICASSP (pp. 4101\u20134104)."},{"key":"9190_CR64","first-page":"433","volume-title":"Proc. of ICASSP","author":"S. E. Tranter","year":"2004","unstructured":"Tranter, S. E., Yu, K., Evermann, G., & Woodland, P. C. (2004). Generating and evaluating segmentations for automatic speech recognition of conversational telephone speech. In Proc. of ICASSP, Montreal, Canada (pp. 433\u2013477)."},{"key":"9190_CR65","doi-asserted-by":"crossref","first-page":"679","DOI":"10.21437\/Eurospeech.1999-174x","volume-title":"EuroSpeech","author":"A. Tritschler","year":"1999","unstructured":"Tritschler, A., & Gopinath, R. (1999). Improved speaker segmentation and segment clustering using the Bayesian information criterion. In EuroSpeech (pp. 679\u2013682)."},{"issue":"4","key":"9190_CR66","doi-asserted-by":"crossref","first-page":"1461","DOI":"10.1109\/TASL.2007.894525","volume":"15","author":"W. H. Tsai","year":"2007","unstructured":"Tsai, W. H., Cheng, S. S., & Wang, H. M. (2007). Automatic speaker clustering using a voice characteristic reference space and maximum purity estimation. IEEE Transactions on Audio, Speech, and Language Processing, 15(4), 1461\u20131474.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"9190_CR67","volume-title":"Speaker odyssey","author":"F. Valente","year":"2004","unstructured":"Valente, F., & Wellekens, C. (2004). Variational Bayesian speaker clustering. In Speaker odyssey, Toledo, Spain."},{"key":"9190_CR68","volume-title":"Proc. of ICASSP","author":"F. Valente","year":"2005","unstructured":"Valente, F., & Wellekens, C. (2005). Variational Bayesian adaptation for speaker clustering. In Proc. of ICASSP, Lisbon, Portugal."},{"key":"9190_CR69","first-page":"4954","volume-title":"ICASSP","author":"F. Valente","year":"2010","unstructured":"Valente, F., Motlicek, P., & Vijayasenan, D. (2010). Variational Bayesian speaker diarization of meeting recordings. In ICASSP (pp. 4954\u20134957)."},{"key":"9190_CR70","first-page":"468","volume-title":"Proc. of ICASSP","author":"D. Wang","year":"2003","unstructured":"Wang, D., Lu, L., & Zhang, H. J. (2003). Speech segmentation without speech recognition. In Proc. of ICASSP, Hong Kong (Vol.\u00a01, pp. 468\u2013471)."},{"key":"9190_CR71","first-page":"555","volume-title":"Lecture notes in computer science","author":"W. Wang","year":"2007","unstructured":"Wang, W., Lv, P., Zhao, Q., & Yan, Y. (2007). A decision-tree-based online speaker clustering. In Lecture notes in computer science (Vol.\u00a04477, pp. 555\u2013562). Berlin: Springer."},{"key":"9190_CR72","doi-asserted-by":"crossref","first-page":"1261","DOI":"10.21437\/Eurospeech.2001-327","volume-title":"Proc. eurospeech","author":"J. Wu","year":"2001","unstructured":"Wu, J., & Chang, E. (2001). Cohorts based custom models for rapid speaker and dialect adaptation. In Proc. eurospeech (pp. 1261\u20131264)."},{"key":"9190_CR73","first-page":"4962","volume-title":"ICASSP","author":"M. Zamalloa","year":"2010","unstructured":"Zamalloa, M., et al. (2010). Low latency online speaker tracking on the AMI corpus of meeting conversations. In ICASSP (pp. 4962\u20134965)."},{"key":"9190_CR74","first-page":"2186","volume-title":"Interspeech","author":"J. Zdansky","year":"2006","unstructured":"Zdansky, J. (2006). BINSEG: an efficient speaker-based segmentation technique. In Interspeech, Pennsylvania (pp. 2186\u20132189)."},{"key":"9190_CR75","isbn-type":"print","doi-asserted-by":"crossref","DOI":"10.1007\/0-387-29151-2","volume-title":"Advances in database systems","author":"P. Zezula","year":"2006","unstructured":"Zezula, P., Amato, G., Dohnal, V., & Batko, M. (2006). Similarity search: the metric space approach. In Advances in database systems (Vol.\u00a032). ISBN 0-387-29146-6","ISBN":"https:\/\/id.crossref.org\/isbn\/0387291466"},{"key":"9190_CR76","first-page":"554","volume-title":"Interspeech","author":"B. Zhou","year":"2002","unstructured":"Zhou, B., & Hansen, J. (2002). Improved structural maximum likelihood eigenspace mapping for rapid speaker adaptation. In Interspeech, Denver, Colorado (pp. 554\u2013564)."},{"issue":"4","key":"9190_CR77","doi-asserted-by":"crossref","first-page":"467","DOI":"10.1109\/TSA.2005.845790","volume":"13","author":"B. Zhou","year":"2005","unstructured":"Zhou, B., & Hansen, J. H. L. (2005). Efficient audio stream segmentation via the combined T2 statistic and the Bayesian information criterion. IEEE Transactions on Speech and Audio Processing, 13(4), 467\u2013474.","journal-title":"IEEE Transactions on Speech and Audio Processing"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-013-9190-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10772-013-9190-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-013-9190-8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,29]],"date-time":"2025-04-29T21:11:32Z","timestamp":1745961092000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10772-013-9190-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013,2,14]]},"references-count":77,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2013,12]]}},"alternative-id":["9190"],"URL":"https:\/\/doi.org\/10.1007\/s10772-013-9190-8","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"type":"print","value":"1381-2416"},{"type":"electronic","value":"1572-8110"}],"subject":[],"published":{"date-parts":[[2013,2,14]]}}}