{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,8]],"date-time":"2026-08-08T07:07:50Z","timestamp":1786172870952,"version":"3.56.0"},"publisher-location":"Berlin, Heidelberg","reference-count":36,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"value":"9783540874782","type":"print"},{"value":"9783540874799","type":"electronic"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"DOI":"10.1007\/978-3-540-87479-9_48","type":"book-chapter","created":{"date-parts":[[2008,8,13]],"date-time":"2008-08-13T19:26:48Z","timestamp":1218655608000},"page":"457-472","source":"Crossref","is-referenced-by-count":16,"title":["Watch, Listen &amp; Learn: Co-training on Captioned Images and Videos"],"prefix":"10.1007","author":[{"given":"Sonal","family":"Gupta","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Joohyun","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kristen","family":"Grauman","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Raymond","family":"Mooney","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","reference":[{"key":"48_CR1","volume-title":"Modern Information Retrieval","author":"R. Baeza-Yates","year":"1999","unstructured":"Baeza-Yates, R., Ribeiro-Neto, B.: Modern Information Retrieval. ACM Press, New York (1999)"},{"key":"48_CR2","doi-asserted-by":"publisher","first-page":"1107","DOI":"10.1162\/153244303322533214","volume":"3","author":"K. Barnard","year":"2003","unstructured":"Barnard, K., Duygulu, P., Forsyth, D., de Freitas, N., Blei, D.M., Jordan, M.I.: Matching words and pictures. Journal of Machine Learning Research\u00a03, 1107\u20131135 (2003)","journal-title":"Journal of Machine Learning Research"},{"key":"48_CR3","volume-title":"Proceedings of the IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR 2007)","author":"R. Bekkerman","year":"2007","unstructured":"Bekkerman, R., Jeon, J.: Multi-modal clustering for multimedia collections. In: Proceedings of the IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR 2007). IEEE Computer Society, Los Alamitos (2007)"},{"key":"48_CR4","first-page":"368","volume":"11","author":"K. Bennett","year":"1999","unstructured":"Bennett, K., Demiriz, A.: Semi-supervised support vector machines. Advances in Neural Information Processing Systems\u00a011, 368\u2013374 (1999)","journal-title":"Advances in Neural Information Processing Systems"},{"key":"48_CR5","doi-asserted-by":"crossref","unstructured":"Blank, M., Gorelick, L., Shechtman, E., Irani, M., Basri, R.: Actions as space-time shapes. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV 2005), pp. 1395\u20131402 (2005)","DOI":"10.1109\/ICCV.2005.28"},{"key":"48_CR6","doi-asserted-by":"crossref","unstructured":"Blum, A., Mitchell, T.: Combining labeled and unlabeled data with co-training. In: Proceedings of the 11th Annual Conference on Computational Learning Theory, Madison, WI, pp. 92\u2013100 (1998)","DOI":"10.1145\/279943.279962"},{"issue":"1","key":"48_CR7","doi-asserted-by":"publisher","first-page":"330","DOI":"10.1016\/j.patcog.2006.06.005","volume":"40","author":"J. Cheng","year":"2007","unstructured":"Cheng, J., Wang, K.: Active learning for image retrieval with Co-SVM. Pattern Recognition\u00a040(1), 330\u2013334 (2007)","journal-title":"Pattern Recognition"},{"key":"48_CR8","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1145\/1180995.1181013","volume-title":"ICMI 2006: Proceedings of the 8th international conference on Multimodal interfaces","author":"C.M. Christoudias","year":"2006","unstructured":"Christoudias, C.M., Saenko, K., Morency, L.-P., Darrell, T.: Co-adaptation of audio-visual speech and gesture classifiers. In: ICMI 2006: Proceedings of the 8th international conference on Multimodal interfaces, pp. 84\u201391. ACM, New York (2006)"},{"key":"48_CR9","doi-asserted-by":"publisher","first-page":"886","DOI":"10.1109\/CVPR.2005.177","volume-title":"Proceedings of the 2005 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR 2005)","author":"N. Dalal","year":"2005","unstructured":"Dalal, N., Triggs, B.: Histograms of oriented gradients for human detection. In: Proceedings of the 2005 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR 2005), Washington, DC, USA, vol.\u00a01, pp. 886\u2013893. IEEE Computer Society, Los Alamitos (2005)"},{"key":"48_CR10","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"97","DOI":"10.1007\/3-540-47979-1_7","volume-title":"Computer Vision - ECCV 2002","author":"P. Duygulu","year":"2002","unstructured":"Duygulu, P., Barnard, K., de Freitas, N., Forsyth, D.: Object recognition as machine translation: Learning a lexicon for a fixed image vocabulary. In: Heyden, A., Sparr, G., Nielsen, M., Johansen, P. (eds.) ECCV 2002. LNCS, vol.\u00a02353, pp. 97\u2013112. Springer, Heidelberg (2002)"},{"key":"48_CR11","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"crossref","first-page":"132","DOI":"10.1007\/978-3-540-27814-6_19","volume-title":"Image and Video Retrieval","author":"P. Duygulu","year":"2004","unstructured":"Duygulu, P., Hauptmann, A.G.: What\u2019s news, what\u2019s not? associating news videos with words. In: Enser, P.G.B., Kompatsiaris, Y., O\u2019Connor, N.E., Smeaton, A.F., Smeulders, A.W.M. (eds.) CIVR 2004. LNCS, vol.\u00a03115, pp. 132\u2013140. Springer, Heidelberg (2004)"},{"key":"48_CR12","doi-asserted-by":"crossref","unstructured":"Efros, A.A., Berg, A.C., Mori, G., Malik, J.: Recognizing action at a distance. In: IEEE International Conference on Computer Vision, Nice, France, pp. 726\u2013733 (2003)","DOI":"10.1109\/ICCV.2003.1238420"},{"key":"48_CR13","doi-asserted-by":"crossref","unstructured":"Everingham, M., Sivic, J., Zisserman, A.: Hello! My name is... Buffy \u2013 Automatic naming of characters in TV video. In: Proceedings of the British Machine Vision Conference (2006)","DOI":"10.5244\/C.20.92"},{"key":"48_CR14","doi-asserted-by":"publisher","first-page":"178","DOI":"10.1109\/CVPR.2004.383","volume-title":"Proceedings of the 2004 Conference on Computer Vision and Pattern Recognition Workshop (CVPRW 2004)","author":"L. Fei-Fei","year":"2004","unstructured":"Fei-Fei, L., Fergus, R., Perona, P.: Learning generative visual models from few training examples: An incremental bayesian approach tested on 101 object categories. In: Proceedings of the 2004 Conference on Computer Vision and Pattern Recognition Workshop (CVPRW 2004), Washington, DC, USA, vol.\u00a012, p. 178. IEEE Computer Society, Los Alamitos (2004)"},{"key":"48_CR15","doi-asserted-by":"crossref","unstructured":"Fleischman, M., Roy, D.: Situated models of meaning for sports video retrieval. In: Human Language Technologies 2007: The Conference of the North American Chapter of the Association for Computational Linguistics, Rochester, New York, April 2007, pp. 37\u201340. Association for Computational Linguistics (2007)","DOI":"10.3115\/1614108.1614118"},{"key":"48_CR16","unstructured":"Forstner, W., Gulch, E.: A fast operator for detection and precise location of distinct points, corners and centres of circular features. In: ISPRS Intercommission Conference on Fast Processing of Photogrammetric Data, pp. 281\u2013305 (1987)"},{"key":"48_CR17","unstructured":"Griffin, G., Holub, A., Perona, P.: Caltech-256 object category dataset. Technical Report 7694, California Institute of Technology (2007)"},{"key":"48_CR18","doi-asserted-by":"crossref","unstructured":"Harris, C., Stephens, M.: A combined corner and edge detector. In: Alvey Vision Conference, pp. 147\u2013152 (1988)","DOI":"10.5244\/C.2.23"},{"key":"48_CR19","unstructured":"Joachims, T.: Transductive inference for text classification using support vector machines. In: Proceedings of the Sixteenth International Conference on Machine Learning (ICML 1999), Bled, Slovenia, June 1999, pp. 200\u2013209 (1999)"},{"key":"48_CR20","doi-asserted-by":"crossref","unstructured":"Ke, Y., Sukthankar, R., Hebert, M.: Event detection in crowded videos. In: IEEE International Conference on Computer Vision (October 2007)","DOI":"10.1109\/ICCV.2007.4409011"},{"key":"48_CR21","unstructured":"Kiritchenko, S., Matwin, S.: Email classification with co-training. In: Proceedings of CASCON 2001, Toronto, Canada, pp. 192\u2013201 (2001)"},{"issue":"2-3","key":"48_CR22","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1007\/s11263-005-1838-7","volume":"64","author":"I. Laptev","year":"2005","unstructured":"Laptev, I.: On space-time interest points. International Journal of Computer Vision\u00a064(2-3), 107\u2013123 (2005)","journal-title":"International Journal of Computer Vision"},{"key":"48_CR23","doi-asserted-by":"crossref","first-page":"626","DOI":"10.1109\/ICCV.2003.1238406","volume-title":"Proceedings of the IEEE International Conference on Computer Vision (ICCV 2003)","author":"A. Levin","year":"2003","unstructured":"Levin, A., Viola, P., Freund, Y.: Unsupervised improvement of visual detectors using co-training. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV 2003), p. 626. IEEE Computer Society, Los Alamitos (2003)"},{"key":"48_CR24","unstructured":"McCallum, A., Nigam, K.: A comparison of event models for naive Bayes text classification. In: Papers from the AAAI 1998 Workshop on Text Categorization, Madison, WI, July 1998, pp. 41\u201348 (1998)"},{"issue":"1","key":"48_CR25","doi-asserted-by":"publisher","first-page":"63","DOI":"10.1023\/B:VISI.0000027790.02288.f2","volume":"60","author":"K. Mikolajczyk","year":"2004","unstructured":"Mikolajczyk, K., Schmid, C.: Scale & affine invariant interest point detectors. International Journal of Computer Vision (IJCV 2004)\u00a060(1), 63\u201386 (2004)","journal-title":"International Journal of Computer Vision (IJCV 2004)"},{"key":"48_CR26","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1023\/A:1007692713085","volume":"39","author":"K. Nigam","year":"2000","unstructured":"Nigam, K., McCallum, A.K., Thrun, S., Mitchell, T.: Text classification from labeled and unlabeled documents using EM. Machine Learning\u00a039, 103\u2013134 (2000)","journal-title":"Machine Learning"},{"key":"48_CR27","doi-asserted-by":"crossref","unstructured":"Nigam, K., Ghani, R.: Analyzing the effectiveness and applicability of co-training. In: Proceedings of the Ninth International Conference on Information and Knowledge Management (CIKM 2000), pp. 86\u201393 (2000)","DOI":"10.1145\/354756.354805"},{"key":"48_CR28","first-page":"4718","volume-title":"ICPR 2000: Proceedings of the International Conference on Pattern Recognition","author":"N. Nitta","year":"2000","unstructured":"Nitta, N., Babaguchi, N., Kitahashi, T.: Extracting actors, actions and events from sports video - a fundamental approach to story tracking. In: ICPR 2000: Proceedings of the International Conference on Pattern Recognition, Washington, DC, USA, p. 4718. IEEE Computer Society, Los Alamitos (2000)"},{"key":"48_CR29","first-page":"61","volume-title":"Advances in Large Margin Classifiers","author":"J.C. Platt","year":"1999","unstructured":"Platt, J.C.: Probabilistic outputs for support vector machines and comparisons to regularized likelihood methods. In: Bartlett, P.J., Sch\u00f6lkopf, B., Schuurmans, D., Smola, A.J. (eds.) Advances in Large Margin Classifiers, pp. 61\u201374. MIT Press, Boston (1999)"},{"key":"48_CR30","first-page":"1","volume-title":"Proceedings of the IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR 2007)","author":"A. Quattoni","year":"2007","unstructured":"Quattoni, A., Collins, M., Darrell, T.: Learning visual representations using images with captions. In: Proceedings of the IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR 2007), June 2007, pp. 1\u20138. IEEE CS Press, Los Alamitos (2007)"},{"key":"48_CR31","doi-asserted-by":"crossref","unstructured":"Rosenberg, C., Hebert, M., Schneiderman, H.: Semi-supervised self-training of object detection models. In: Proceedings of the Seventh IEEE Workshops on Application of Computer Vision (WACV\/MOTION 2005), Washington, DC, USA, vol.\u00a01, pp. 29\u201336. IEEE Computer Society, Los Alamitos (2005)","DOI":"10.1109\/ACVMOT.2005.107"},{"key":"48_CR32","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1109\/ICPR.2004.1334462","volume-title":"Proceedings of the Pattern Recognition, 17th International Conference on (ICPR 2004)","author":"C. Schuldt","year":"2004","unstructured":"Schuldt, C., Laptev, I., Caputo, B.: Recognizing human actions: A local SVM approach. In: Proceedings of the Pattern Recognition, 17th International Conference on (ICPR 2004), Washington, DC, USA, vol.\u00a03, pp. 32\u201336. IEEE Computer Society, Los Alamitos (2004)"},{"key":"48_CR33","doi-asserted-by":"publisher","first-page":"399","DOI":"10.1145\/1101149.1101236","volume-title":"MULTIMEDIA 2005: Proceedings of the 13th annual ACM international conference on Multimedia","author":"C.G.M. Snoek","year":"2005","unstructured":"Snoek, C.G.M., Worring, M., Smeulders, A.W.M.: Early versus late fusion in semantic video analysis. In: MULTIMEDIA 2005: Proceedings of the 13th annual ACM international conference on Multimedia, pp. 399\u2013402. ACM, New York (2005)"},{"key":"48_CR34","doi-asserted-by":"crossref","unstructured":"Wang, J., Duan, L., Xu, L., Lu, H., Jin, J.S.: Tv ad video categorization with probabilistic latent concept learning. In: Multimedia Information Retrieval, pp. 217\u2013226 (2007)","DOI":"10.1145\/1290082.1290113"},{"key":"48_CR35","doi-asserted-by":"crossref","unstructured":"Wang, Y., Sabzmeydani, P., Mori, G.: Semi-latent Dirichlet allocation: A hierarchical model for human action recognition. In: 2nd Workshop on Human Motion Understanding, Modeling, Capture and Animation (2007)","DOI":"10.1007\/978-3-540-75703-0_17"},{"key":"48_CR36","volume-title":"Data Mining: Practical Machine Learning Tools and Techniques with Java Implementations","author":"I.H. Witten","year":"2005","unstructured":"Witten, I.H., Frank, E.: Data Mining: Practical Machine Learning Tools and Techniques with Java Implementations, 2nd edn. Morgan Kaufman Publishers, San Francisco (2005)","edition":"2"}],"container-title":["Lecture Notes in Computer Science","Machine Learning and Knowledge Discovery in Databases"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-540-87479-9_48.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,11,24]],"date-time":"2020-11-24T02:38:07Z","timestamp":1606185487000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-540-87479-9_48"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[null]]},"ISBN":["9783540874782","9783540874799"],"references-count":36,"URL":"https:\/\/doi.org\/10.1007\/978-3-540-87479-9_48","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[]}}