{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T10:10:45Z","timestamp":1775470245692,"version":"3.50.1"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2016,10,20]],"date-time":"2016-10-20T00:00:00Z","timestamp":1476921600000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2016,12]]},"DOI":"10.1007\/s10772-016-9384-y","type":"journal-article","created":{"date-parts":[[2016,10,20]],"date-time":"2016-10-20T09:28:03Z","timestamp":1476955683000},"page":"945-963","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":12,"title":["Speaker diarization system using MKMFCC parameterization and WLI-fuzzy clustering"],"prefix":"10.1007","volume":"19","author":[{"given":"V. Subba","family":"Ramaiah","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"R. Rajeswara","family":"Rao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,10,20]]},"reference":[{"issue":"8","key":"9384_CR1","doi-asserted-by":"crossref","first-page":"649","DOI":"10.1109\/LSP.2004.831666","volume":"11","author":"J Ajmera","year":"2004","unstructured":"Ajmera, J., McCowan, I., & Bourlard, H. (2004). Robust Speaker Change Detection. IEEE Signal Processing Letters, 11(8), 649\u2013651.","journal-title":"IEEE Signal Processing Letters"},{"key":"9384_CR2","unstructured":"Anguera, X., Bozonnet, S., Evans, N., Fredouille, C., & Friedland, G. (2010). Speaker diarization: A review of recent research. In Proceedings of IEEE TASLP (pp. 1\u201314)."},{"key":"9384_CR3","unstructured":"Bakis, R., Chen, S., Gopalakrishnan, P., Gopinath, R., Maes, S., Polymenakos, L., & Franz, M. (1997). Transcription of broadcast news shows with the IBM large vocabulary speech recognition system. In Proceedings of the speech recognition workshop (pp. 67\u201372)."},{"issue":"5","key":"9384_CR4","doi-asserted-by":"crossref","first-page":"1505","DOI":"10.1109\/TASL.2006.878261","volume":"14","author":"C Barras","year":"2006","unstructured":"Barras, C., Zhu, X., Meignier, S., & Gauvain, J.-L. (2006). Multistage Speaker Diarization of Broadcast News. IEEE Transactions on Audio, Speech and Language Processing, 14(5), 1505\u20131512.","journal-title":"IEEE Transactions on Audio, Speech and Language Processing"},{"key":"9384_CR5","unstructured":"Beigi, H., & Maes, S. (1998). Speaker, channel and environment change detection. In Proceedings of the world congress on automation."},{"issue":"2\u20133","key":"9384_CR6","doi-asserted-by":"crossref","first-page":"191","DOI":"10.1016\/0098-3004(84)90020-7","volume":"10","author":"JC Bezdek","year":"1984","unstructured":"Bezdek, J. C., Ehrlich, R., & Full, W. (1984). FCM: The fuzzy c-means clustering algorithm. Computers & Geosciences, 10(2\u20133), 191\u2013203.","journal-title":"Computers & Geosciences"},{"key":"9384_CR7","first-page":"127","volume":"8","author":"S Chen","year":"1998","unstructured":"Chen, S., & Gopalakrishnan, P. (1998). Speaker, environment and channel change detection and clustering via the bayesian information criterion. Proceedings of DARPA Broadcast News Transcription and Understanding Workshop, 8, 127\u2013132.","journal-title":"Proceedings of DARPA Broadcast News Transcription and Understanding Workshop"},{"issue":"7","key":"9384_CR8","doi-asserted-by":"crossref","first-page":"1023","DOI":"10.1016\/S0165-1684(02)00206-2","volume":"82","author":"J Chen","year":"2002","unstructured":"Chen, J., Shue, L., & Ser, W. (2002). A new approach for speaker tracking in reverberant environment. Signal Processing, 82(7), 1023\u20131028.","journal-title":"Signal Processing"},{"key":"9384_CR39","unstructured":"CSTR VCTK Corpus from http:\/\/homepages.inf.ed.ac.uk\/jyamagis\/page3\/page58\/page58.html"},{"issue":"4","key":"9384_CR9","doi-asserted-by":"crossref","first-page":"357","DOI":"10.1109\/TASSP.1980.1163420","volume":"28","author":"S Davis","year":"1980","unstructured":"Davis, S., & Mermelstein, P. (1980). Comparison of parametric representations for monosyllabic word recognition in continuously spoken sentences. IEEE Transaction on Acoustic Speech Signal Processing, 28(4), 357\u2013366.","journal-title":"IEEE Transaction on Acoustic Speech Signal Processing"},{"issue":"12","key":"9384_CR10","doi-asserted-by":"crossref","first-page":"2286","DOI":"10.1109\/TASLP.2015.2479043","volume":"23","author":"H Delgado","year":"2015","unstructured":"Delgado, H., Anguera, X., Fredouille, C., & Serrano, J. (2015). Fast single- and cross-show speaker diarization using binary key speaker modeling. IEEE Transactions on Audio, Speech and Language Processing, 23(12), 2286\u20132297.","journal-title":"IEEE Transactions on Audio, Speech and Language Processing"},{"issue":"1\u20133","key":"9384_CR11","doi-asserted-by":"crossref","first-page":"93","DOI":"10.1006\/dspr.1999.0359","volume":"10","author":"RB Dunn","year":"2000","unstructured":"Dunn, R. B., Reynolds, D. A., & Quatieri, T. F. (2000). Approaches to speaker detection and tracking in conversational speech. Digital Signal Processing, 10(1\u20133), 93\u2013112.","journal-title":"Digital Signal Processing"},{"key":"9384_CR38","unstructured":"ELSDSR database from http:\/\/cogsys.compute.dtu.dk\/soundshare\/elsdsr.zip"},{"issue":"2","key":"9384_CR12","doi-asserted-by":"crossref","first-page":"382","DOI":"10.1109\/TASL.2011.2159710","volume":"20","author":"N Evans","year":"2012","unstructured":"Evans, N., Bozonnet, S., & Wang, D. (2012). A comparative study of bottom-up and top-down approaches to speaker diarization. IEEE Transactions on Audio, Speech and Language Processing, 20(2), 382\u2013392.","journal-title":"IEEE Transactions on Audio, Speech and Language Processing"},{"issue":"4","key":"9384_CR13","doi-asserted-by":"crossref","first-page":"18","DOI":"10.1109\/79.317924","volume":"11","author":"H Gish","year":"1994","unstructured":"Gish, H., & Schmidt, N. (1994). Text-independent speaker identification. IEEE Signal Processing Magazine, 11(4), 18\u201332.","journal-title":"IEEE Signal Processing Magazine"},{"issue":"2","key":"9384_CR14","doi-asserted-by":"crossref","first-page":"404","DOI":"10.1109\/TASL.2011.2162320","volume":"20","author":"M Huijbregts","year":"2012","unstructured":"Huijbregts, M., & van Leeuwen, D. A. (2012). Large-scale speaker diarization for long recordings and small collections. IEEE Transaction on Audio, Speech, and Language Processing, 20(2), 404\u2013413.","journal-title":"IEEE Transaction on Audio, Speech, and Language Processing"},{"issue":"3","key":"9384_CR15","doi-asserted-by":"crossref","first-page":"335","DOI":"10.1109\/TKDE.2010.122","volume":"23","author":"J-Y Jiang","year":"2011","unstructured":"Jiang, J.-Y., Liou, R.-J., & Lee, S.-J. (2011). A fuzzy self-constructing feature clustering algorithm for text classification. IEEE Transactions on Knowledge and Data Engineering, 23(3), 335\u2013349.","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"9384_CR16","unstructured":"Kenny, P., Gupta, V., Stafylakis, T., Ouellet, P. & Alam, J. (2014). Deep neural networks for Baum-Wech statistics for speaker Recognition. In Proceedings of neural networks for speaker and language modelling."},{"key":"9384_CR17","unstructured":"Kubala, F., Jin, H., Matsoukas, S., Nguyen, L., Schwartz, R., & Makhou, J. (1997). The 1996 BBN Byblos Hub-4 transcription system. In Proceedings of the speech recognition workshop (pp. 90\u201393)."},{"key":"9384_CR18","first-page":"1869","volume":"7","author":"VB Le","year":"2007","unstructured":"Le, V. B., Mella, O., & Fohr, D. (2007). Speaker diarization using normalized cross likelihood ratio. Interspeech, 7, 1869\u20131872.","journal-title":"Interspeech"},{"key":"9384_CR19","doi-asserted-by":"crossref","unstructured":"Madikeri, S., Himawan, I., Motlicek, P., & Ferras, M. (2015). Integrating online i-vector extractor with information bottleneck based speaker diarization system. In Interspeech-2015 16 th Annual Conference of the international speech communication association, Dresden, Germany, September 6\u201310 (pp. 3105\u20133109).","DOI":"10.21437\/Interspeech.2015-111"},{"issue":"4","key":"9384_CR20","doi-asserted-by":"crossref","first-page":"561","DOI":"10.1109\/PROC.1975.9792","volume":"63","author":"J Makhoul","year":"1975","unstructured":"Makhoul, J. (1975). Linear prediction: a tutorial review. Proceedings of IEEE, 63(4), 561\u2013580.","journal-title":"Proceedings of IEEE"},{"key":"9384_CR21","unstructured":"Meignier, S., Bonastre, J.-F., & Igounet, S. (2001). E-HMM approach for learning and adapting sound models for speaker indexing. In Proceedings of Odyssey workshop (pp. 175\u2013180)."},{"issue":"6","key":"9384_CR22","doi-asserted-by":"crossref","first-page":"763","DOI":"10.1016\/j.specom.2012.01.005","volume":"54","author":"MH Moattar","year":"2012","unstructured":"Moattar, M. H., & Homayounpour, M. M. (2012). Variational conditional random fields for online speaker detection and tracking. Speech Communication, 54(6), 763\u2013780.","journal-title":"Speech Communication"},{"issue":"3","key":"9384_CR23","first-page":"138","volume":"2","author":"L Muda","year":"2010","unstructured":"Muda, L., Begam, M., & Elamvazuthi, I. (2010). Voice recognition algorithms using Mel frequency cepstral coefficient (MFCC) and dynamic time warping (DTW) techniques. Journal of Computing, 2(3), 138\u2013153.","journal-title":"Journal of Computing"},{"key":"9384_CR24","doi-asserted-by":"crossref","first-page":"229","DOI":"10.1007\/978-1-4939-1456-2_8","volume-title":"Speech and audio processing for coding, enhancement and recognition, Part II","author":"TH Nguyen","year":"2015","unstructured":"Nguyen, T. H., Chng, E. S., & Li, H. (2015). Speaker diarization: An emerging research. In T. Ogunfunmi (Ed.), Speech and audio processing for coding, enhancement and recognition, Part II (pp. 229\u2013277). New York: Springer."},{"key":"9384_CR25","unstructured":"NIST. (2009). The NIST Rich Transcription 2009 (RT\u201909) evaluation. http:\/\/www.itl.nist.gov\/iad\/mig\/tests\/rt\/2009\/docs\/rt09-meeting-val-plan-v2.pdf ."},{"key":"9384_CR26","doi-asserted-by":"crossref","unstructured":"Oku, T., Sato, S., Kobayashi, A., Homma, S., & Imai, T. (2012). Low-latency speaker diarization based on Bayesian information criterion with multiple phoneme classes. In Proceedings of IEEE international conference on acoustics, speech and signal processing (pp. 4189\u20134192).","DOI":"10.1109\/ICASSP.2012.6288842"},{"issue":"3","key":"9384_CR27","doi-asserted-by":"crossref","first-page":"683","DOI":"10.1016\/j.csl.2012.08.003","volume":"27","author":"P Pertila","year":"2013","unstructured":"Pertila, P. (2013). Online blind speech separation using multiple acoustic speaker tracking and time\u2013frequency masking. Computer Speech & Language, 27(3), 683\u2013702.","journal-title":"Computer Speech & Language"},{"key":"9384_CR40","unstructured":"Ramaiah, V. S., & Rao, R. R. (in press) A novel approach for speaker diarization system using tmfcc parameterization and lion optimization. Journal of Central South University of Technology."},{"key":"9384_CR28","doi-asserted-by":"crossref","unstructured":"Reynolds, D. (2009). Universal background models. In Encyclopedia of biometrics (pp. 1349\u20131352). New York: Springer.","DOI":"10.1007\/978-0-387-73003-5_197"},{"issue":"1","key":"9384_CR29","doi-asserted-by":"crossref","first-page":"72","DOI":"10.1109\/89.365379","volume":"3","author":"DA Reynolds","year":"1995","unstructured":"Reynolds, D. A., & Rose, R. C. (1995). Robust text-independent speaker identification using Gaussian mixture speaker models. IEEE Transactions on Speech and Audio Processing, 3(1), 72\u201383.","journal-title":"IEEE Transactions on Speech and Audio Processing"},{"key":"9384_CR30","unstructured":"Siegler, M. A., Jain, U., Raj, B., & Stern, R. M. (1997). Automatic segmentation, classification and clustering of broadcast news audio. In Proceedings of DARPA speech recognition workshop (pp. 97\u201399)."},{"key":"9384_CR31","unstructured":"Siegler, M., Jain, U., Ray, B., & Stern, R. (1997). Automatic segmentation, classifcation and clustering of broadcast news audio. In Proceedings of the speech recognition workshop (pp. 97\u201399)."},{"key":"9384_CR32","first-page":"84","volume":"4","author":"JS Sohal","year":"2015","unstructured":"Sohal, J. S., & Sukhvinder, K. (2015). Optimization of speaker diarization by reducing diarization error rate: A review. International Journal of Electronics and Communication Engineering, 4, 84\u201387.","journal-title":"International Journal of Electronics and Communication Engineering"},{"issue":"3","key":"9384_CR33","doi-asserted-by":"crossref","first-page":"329","DOI":"10.2307\/1417526","volume":"53","author":"S Stevens","year":"1940","unstructured":"Stevens, S., & Volkmann, J. (1940). The relation of pitch to frequency: a revised scale. The American Journal of Psychology, 53(3), 329\u2013353.","journal-title":"The American Journal of Psychology"},{"key":"9384_CR34","doi-asserted-by":"crossref","first-page":"55","DOI":"10.1016\/j.specom.2011.07.001","volume":"54","author":"D Vijayasenan","year":"2012","unstructured":"Vijayasenan, D., Valente, F., & Bourlard, H. (2012). Multistream speaker diarization of meetings recordings beyond MFCC and TDOA features. Speech Communication, 54, 55\u201367.","journal-title":"Speech Communication"},{"key":"9384_CR35","unstructured":"Woodland, P., Gales, M., Pye, D., & Young, S. (1997). The development of the 1996 HTK broadcast news transcription system. In Proceedings of the speech recognition workshop (pp. 73\u201378)."},{"issue":"3","key":"9384_CR36","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TFUZZ.2012.2197754","volume":"23","author":"C-H Wu","year":"2013","unstructured":"Wu, C.-H., Ouyang, C.-S., Chen, L.-W., & Lu, L.-W. (2013). A new fuzzy clustering validity index with a median factor for centroid-based clustering. IEEE Transaction on Fuzzy System, 23(3), 1\u201316.","journal-title":"IEEE Transaction on Fuzzy System"},{"issue":"9","key":"9384_CR37","doi-asserted-by":"crossref","first-page":"3393","DOI":"10.1007\/s00034-015-0206-2","volume":"35","author":"Y Xu","year":"2015","unstructured":"Xu, Y., McLoughlin, I., Song, Y., & Wu, K. (2015). Improved i-vector representation for speaker diarization. Circuits, Systems, and Signal Processing, 35(9), 3393\u20133404.","journal-title":"Circuits, Systems, and Signal Processing"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-016-9384-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10772-016-9384-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-016-9384-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,6,20]],"date-time":"2024-06-20T02:26:13Z","timestamp":1718850373000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10772-016-9384-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,10,20]]},"references-count":40,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2016,12]]}},"alternative-id":["9384"],"URL":"https:\/\/doi.org\/10.1007\/s10772-016-9384-y","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2016,10,20]]}}}