{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,11]],"date-time":"2025-07-11T10:30:13Z","timestamp":1752229813055},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2014,5,9]],"date-time":"2014-05-09T00:00:00Z","timestamp":1399593600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2014,9]]},"DOI":"10.1007\/s13735-014-0055-y","type":"journal-article","created":{"date-parts":[[2014,5,8]],"date-time":"2014-05-08T12:28:39Z","timestamp":1399552119000},"page":"161-175","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["Person instance graphs for mono-, cross- and multi-modal person recognition in multimedia data: application to speaker identification in TV broadcast"],"prefix":"10.1007","volume":"3","author":[{"given":"Herv\u00e9","family":"Bredin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Anindya","family":"Roy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Viet-Bac","family":"Le","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Claude","family":"Barras","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,5,9]]},"reference":[{"issue":"5","key":"55_CR1","doi-asserted-by":"crossref","first-page":"1505","DOI":"10.1109\/TASL.2006.878261","volume":"14","author":"C Barras","year":"2006","unstructured":"Barras C, Zhu X, Meignier S, Gauvain JL (2006) Multi-stage speaker diarization of broadcast news. IEEE Trans Audio Speech Lang Process 14(5):1505\u20131512","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"55_CR2","doi-asserted-by":"crossref","unstructured":"B\u00e4uml M, Tapaswi M, Stiefelhagen R (2013) Semi-supervised learning with constraints for person identification in multimedia data. In: International conference on computer vision and pattern recognition (CVPR)","DOI":"10.1109\/CVPR.2013.462"},{"key":"55_CR3","first-page":"281","volume":"13","author":"J Bergstra","year":"2012","unstructured":"Bergstra J, Bengio Y (2012) Random search for hyper-parameter optimization. J Mach Learn Res 13:281\u2013305","journal-title":"J Mach Learn Res"},{"key":"55_CR4","doi-asserted-by":"crossref","unstructured":"Blondel VD, Guillaume JL, Lambiotte R, Lefebvre E (2008) Fast unfolding of communities in large networks. J Stat Mech Theory Exp 2008(10):P10008. doi: 10.1088\/1742-5468\/2008\/10\/P10008","DOI":"10.1088\/1742-5468\/2008\/10\/P10008"},{"key":"55_CR5","doi-asserted-by":"crossref","unstructured":"Bredin H, Chollet G (2007) Audio-visual speech synchrony measure: application to biometrics. EURASIP J Adv Signal Process 2007(1):070186. doi: 10.1155\/2007\/70186","DOI":"10.1155\/2007\/70186"},{"key":"55_CR6","doi-asserted-by":"crossref","unstructured":"Bredin H, Poignant J (2013) Integer linear programming for speaker diarization and cross-modal identification in TV broadcast. In: Interspeech 2013, 14th annual conference of the International Speech Communication Association, Lyon","DOI":"10.21437\/Interspeech.2013-381"},{"key":"55_CR7","doi-asserted-by":"crossref","unstructured":"Canseco L, Lamel L, Gauvain JL (2005) A comparative study using manual and automatic transcriptions for diarization. In: Proceedings of the IEEE automatic speech recognition and understanding, workshop, pp 415\u2013419","DOI":"10.1109\/ASRU.2005.1566507"},{"key":"55_CR8","unstructured":"Chen SS, Gopalakrishnan P (1998) Speaker, environment and channel change detection and clustering via the Bayesian information criterion. In: DARPA broadcast news transcription and understanding workshop. Virginia"},{"key":"55_CR9","doi-asserted-by":"crossref","unstructured":"Cour T, Sapp B, Nagle A, Taskar B (2010) Talking pictures: temporal grouping and dialog-supervised person recognition. In: International conference on computer vision and pattern recognition (CVPR)","DOI":"10.1109\/CVPR.2010.5540106"},{"issue":"3","key":"55_CR10","doi-asserted-by":"crossref","first-page":"42","DOI":"10.1109\/MMUL.2002.1022858","volume":"9","author":"N Dimitrova","year":"2002","unstructured":"Dimitrova N, Zhang HJ, Shahraray B, Sezan I, Huang T, Zakhor A (2002) Applications of video-content analysis and retrieval. IEEE Multimed 9(3):42\u201355","journal-title":"IEEE Multimed"},{"key":"55_CR11","unstructured":"Dinarelli M, Rosset S (2011) Models cascade for tree-structured named entity detection. In: Proceedings of 5th international joint conference on natural language processing, Asian Federation of Natural Language processing, Chiang Mai, pp 1269\u20131278"},{"key":"55_CR12","doi-asserted-by":"crossref","unstructured":"Dupuy G, Rouvier M, Meignier S, Est\u00e8ve Y (2012) i-Vectors and ILP clustering adapted to cross-show speaker diarization. In: Interspeech 2012, 13th annual conference of the International Speech Communication Association","DOI":"10.21437\/Interspeech.2012-580"},{"key":"55_CR13","doi-asserted-by":"crossref","unstructured":"Est\u00e8ve Y, Meignier S, Del\u00e9glise, P, Mauclair J (2007) Extracting true speaker identities from transcriptions. In: Proceedings of interspeech, pp 2601\u20132604","DOI":"10.21437\/Interspeech.2007-586"},{"key":"55_CR14","doi-asserted-by":"crossref","unstructured":"Finkel JR, Manning CD (2008) Enforcing transitivity in coreference resolution. In: Annual meeting of the Association for Computational Linguistics: Human Language Technologies (ACL HLT)","DOI":"10.3115\/1557690.1557703"},{"key":"55_CR15","unstructured":"Fiscus JG, Garofolo, JS, Le, AN, Martin, AF, Pallett D, Przybocki MA, Sanders GA (2004) Results of the Fall 2004 STT and MDE evaluation. In: Fall 2004 rich transcription workshop (RT-04). Palisades"},{"key":"55_CR16","doi-asserted-by":"crossref","unstructured":"Gauvain JL, Lamel L, Adda G (1998) Partitioning and transcription of broadcast news data. In: Proceedings of international conference on spoken language processing (ICSLP 98), Sydney, pp 1335\u20131338","DOI":"10.21437\/ICSLP.1998-618"},{"issue":"1\u20132","key":"55_CR17","doi-asserted-by":"crossref","first-page":"89","DOI":"10.1016\/S0167-6393(01)00061-9","volume":"37","author":"JL Gauvain","year":"2002","unstructured":"Gauvain JL, Lamel L, Adda G (2002) The limsi broadcast news transcription system. Speech Commun 37(1\u20132):89\u2013109","journal-title":"Speech Commun"},{"issue":"2","key":"55_CR18","doi-asserted-by":"crossref","first-page":"291","DOI":"10.1109\/89.279278","volume":"2","author":"JL Gauvain","year":"1994","unstructured":"Gauvain JL, Lee CH (1994) Maximum a posteriori estimation for multivariate gaussian mixture observations of markov chains. IEEE Trans Speech Audio Process 2(2):291\u2013298","journal-title":"IEEE Trans Speech Audio Process"},{"key":"55_CR19","unstructured":"Giraudel A, Carr\u00e9 M, Mapelli V, Kahn J, Galibert O, Quintard L (2012) The REPERE corpus: a multimodal corpus for person recognition. In: International conference on language resources and evaluation (LREC)"},{"key":"55_CR20","unstructured":"Gravier G, Adda G, Paulson N, Carr\u00e9 M, Giraudel A, Galibert O (2012) The ETAPE corpus for the evaluation of speech-based TV content processing in the French language. In: International conference on language resources, evaluation and corpora, Turkey"},{"key":"55_CR21","unstructured":"Gurobi Optimization Inc (2012) Gurobi optimizer reference manual. http:\/\/www.gurobi.com . Accessed 5 May 2014"},{"issue":"4","key":"55_CR22","doi-asserted-by":"crossref","first-page":"1738","DOI":"10.1121\/1.399423","volume":"87","author":"H Hermansky","year":"1990","unstructured":"Hermansky H (1990) Perceptual linear predictive (PLP) analysis of speech. J Acoust Soc Am 87(4):1738\u20131752. doi: 10.1121\/1.399423","journal-title":"J Acoust Soc Am"},{"issue":"3","key":"55_CR23","doi-asserted-by":"crossref","first-page":"264","DOI":"10.1145\/331499.331504","volume":"31","author":"AK Jain","year":"1999","unstructured":"Jain AK, Murty MN, Flynn PJ (1999) Data clustering: a review. ACM Comput Surv 31(3):264\u2013323","journal-title":"ACM Comput Surv"},{"key":"55_CR24","doi-asserted-by":"crossref","unstructured":"Jousse V, Petitrenaud S, Meignier S, Est\u00e8ve Y, Jacquin C (2009) Automatic named identification of speakers using diarization and ASR systems. In: ICASSP 2009, IEEE international conference on acoustics, speech, and signal processing, Ta\u00efpei","DOI":"10.1109\/ICASSP.2009.4960644"},{"key":"55_CR25","unstructured":"Lawto J, Gauvain JL, Lamel L, Grefenstette G, Gravier G, Despres J, Guinaudeau C, Sebillot P (2011) A scalable video search engine based on audio content indexing and topic segmentation. In: Networked and electronic media (NEM) summit : implementing future media internet"},{"key":"55_CR26","unstructured":"Le VB, Barras C, Ferras M (2010) On the use of GSV-SVM for speaker diarization and tracking. In: Proceedings of Odyssey 2010\u2014the speaker and language recognition workshop, Brno, pp 146\u2013150"},{"key":"55_CR27","unstructured":"Long B, Zhang MZ, Yu PS, Tianbing X (2008) Clustering on complex graphs. In: Proceedings of the twenty-third AAAI conference on artificial intelligence"},{"key":"55_CR28","doi-asserted-by":"crossref","unstructured":"Mauclair J, Meignier S, Est\u00e8ve Y (2006) Speaker diarization: about whom the speaker is talking? In: IEEE Odyssey","DOI":"10.1109\/ODYSSEY.2006.248114"},{"key":"55_CR29","first-page":"408","volume":"2010","author":"S Mouysset","year":"2011","unstructured":"Mouysset S, Noailles J, Ruiz D, Guivarch R (2011) On a strategy for spectral clustering with parallel computation. High Perform Comput Comput Sci VECPAR 2010:408\u2013420","journal-title":"High Perform Comput Comput Sci VECPAR"},{"issue":"23","key":"55_CR30","doi-asserted-by":"crossref","first-page":"8577","DOI":"10.1073\/pnas.0601602103","volume":"103","author":"MEJ Newman","year":"2006","unstructured":"Newman MEJ (2006) Modularity and community structure in networks. Proc Natl Acad Sci USA 103(23):8577\u20138582","journal-title":"Proc Natl Acad Sci USA"},{"key":"55_CR31","unstructured":"Pan JY, Yang HJ, Faloutsos C (2004) MMSS: Multi-modal story-oriented video summarization. In: Proceedings of the fourth IEEE international conference on data mining (ICDM)"},{"key":"55_CR32","doi-asserted-by":"crossref","unstructured":"Pan JY, Yang HJ, Faloutsos C, Duygulu P (2004) Automatic multimedia cross-modal correlation discovery. In: Proceedings of the 10th ACM SIGKDD conference","DOI":"10.1145\/1014052.1014135"},{"key":"55_CR33","unstructured":"Pelecanos J, Sridharan S (2001) Feature warping for robust speaker verification. In: Proceedings of Odyssey 2001\u2014the speaker recognition workshop, Crete, pp 213\u2013218"},{"key":"55_CR34","unstructured":"Pelleg D, Moore AW (2000) X-means: extending K-means with efficient estimation of the number of clusters. Proceedings of the seventeenth international conference on machine learning, ICML \u201900Morgan Kaufmann Publishers Inc., San Francisco, pp 727\u2013734"},{"key":"55_CR35","doi-asserted-by":"crossref","unstructured":"Poignant J, Besacier L, Le VB, Rosset S, Qu\u00e9not G (2013) Unsupervised naming of speakers in broadcast TV: using written names, pronounced names or both? In: Interspeech 2013, 14th annual conference of the International Speech Communication Association, Lyon","DOI":"10.21437\/Interspeech.2013-380"},{"key":"55_CR36","doi-asserted-by":"crossref","unstructured":"Poignant J, Besacier L, Qu\u00e9not G, Thollard F (2012) From text detection in videos to person identification. In: International conference on multimedia and expo (ICME)","DOI":"10.1109\/ICME.2012.119"},{"key":"55_CR37","doi-asserted-by":"crossref","unstructured":"Poignant J, Bredin H, Le VB, Besacier L, Barras C, Qu\u00e9not G (2012) Unsupervised speaker identification using overlaid texts in TV broadcast. In: Interspeech 2012, 13th annual conference of the International Speech Communication Association, Portland","DOI":"10.21437\/Interspeech.2012-344"},{"issue":"1\u20133","key":"55_CR38","doi-asserted-by":"crossref","first-page":"19","DOI":"10.1006\/dspr.1999.0361","volume":"10","author":"DA Reynolds","year":"2000","unstructured":"Reynolds DA, Quatieri TF, Dunn RB (2000) Speaker verification using adapted gaussian mixture models. Digit Signal Process 10(1\u20133):19\u201341","journal-title":"Digit Signal Process"},{"issue":"12","key":"55_CR39","doi-asserted-by":"crossref","first-page":"1349","DOI":"10.1109\/34.895972","volume":"22","author":"A Smeulders","year":"2000","unstructured":"Smeulders A, Worring M, Santini S, Gupta A, Jain R (2000) Content-based image retrieval at the end of the early years. IEEE Trans Pattern Anal Mach Intell 22(12):1349\u20131380","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"55_CR40","doi-asserted-by":"crossref","unstructured":"Smith R (2007) An overview of the tesseract OCR engine. In: Proceedings of the ninth international conference on document analysis and recognition, vol 02, ICDAR \u201907IEEE Computer Society, Washington, DC, pp 629\u2013633","DOI":"10.1109\/ICDAR.2007.4376991"},{"key":"55_CR41","doi-asserted-by":"crossref","unstructured":"Tranter SE (2006) Who really spoke when? Finding speaker turns and identities in broadcast news audio. In: Proceedings of the ICASSP, pp 1013\u20131016","DOI":"10.1109\/ICASSP.2006.1660195"},{"issue":"6","key":"55_CR42","doi-asserted-by":"crossref","first-page":"12","DOI":"10.1109\/79.888862","volume":"17","author":"Y Wang","year":"2000","unstructured":"Wang Y, Liu Z, Huang JC (2000) Multimedia content analysis-using both audio and visual clues. IEEE Signal Process Mag 17(6):12\u201336","journal-title":"IEEE Signal Process Mag"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-014-0055-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s13735-014-0055-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-014-0055-y","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,12]],"date-time":"2023-07-12T23:29:22Z","timestamp":1689204562000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s13735-014-0055-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,5,9]]},"references-count":42,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2014,9]]}},"alternative-id":["55"],"URL":"https:\/\/doi.org\/10.1007\/s13735-014-0055-y","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"value":"2192-6611","type":"print"},{"value":"2192-662X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,5,9]]}}}