{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,13]],"date-time":"2025-02-13T05:20:20Z","timestamp":1739424020560,"version":"3.37.0"},"reference-count":77,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2009,9,30]],"date-time":"2009-09-30T00:00:00Z","timestamp":1254268800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Machine Vision and Applications"],"published-print":{"date-parts":[[2011,1]]},"DOI":"10.1007\/s00138-009-0217-8","type":"journal-article","created":{"date-parts":[[2009,9,29]],"date-time":"2009-09-29T10:33:27Z","timestamp":1254220407000},"page":"99-115","source":"Crossref","is-referenced-by-count":2,"title":["Multimedia translation for linking visual data to semantics in videos"],"prefix":"10.1007","volume":"22","author":[{"given":"P\u0131nar","family":"Duygulu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Muhammet","family":"Ba\u015ftan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2009,9,30]]},"reference":[{"issue":"12","key":"217_CR1","doi-asserted-by":"crossref","first-page":"1349","DOI":"10.1109\/34.895972","volume":"22","author":"A. Smeulders","year":"2000","unstructured":"Smeulders A., Worring M., Santini S., Gupta A., Jain R.: Content based image retrieval at the end of the early years. IEEE Trans. Pattern Anal. Mach. Intell. 22(12), 1349\u20131380 (2000)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"1","key":"217_CR2","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/1126004.1126005","volume":"2","author":"M.S. Lew","year":"2006","unstructured":"Lew M.S., Sebe N., Djeraba C., Jain R.: Content-based multimedia information retrieval: state of the art and challenges. ACM Trans. Multimedia Comput. Commun. Appl. 2(1), 1\u201319 (2006)","journal-title":"ACM Trans. Multimedia Comput. Commun. Appl."},{"key":"217_CR3","doi-asserted-by":"crossref","unstructured":"Hare, J., Lewis, P., Enser, P., Sandom, C.: Mind the gap: another look at the problem of the semantic gap in image retrieval. In: Multimedia Content Analysis, Management and Retrieval, SPIE, vol. 6073, San Jose, California, USA (2006)","DOI":"10.1117\/12.647755"},{"issue":"5","key":"217_CR4","doi-asserted-by":"crossref","first-page":"431","DOI":"10.1109\/69.166986","volume":"4","author":"S. Chang","year":"1992","unstructured":"Chang S., Hsu A.: Image information systems: where do we go from here?. IEEE Trans. Knowl. Data Eng. 4(5), 431\u2013442 (1992)","journal-title":"IEEE Trans. Knowl. Data Eng."},{"issue":"4","key":"217_CR5","doi-asserted-by":"crossref","first-page":"39","DOI":"10.1006\/jvci.1999.0413","volume":"10","author":"Y. Rui","year":"1999","unstructured":"Rui Y., Huang T., Chang S.: Image retrieval: current techniques, promising directions, and open issues. J. Vis. Commun. Image Represent. 10(4), 39\u201362 (1999)","journal-title":"J. Vis. Commun. Image Represent."},{"issue":"1","key":"217_CR6","doi-asserted-by":"crossref","first-page":"5","DOI":"10.1023\/B:MTAP.0000046380.27575.a5","volume":"25","author":"C. Snoek","year":"2005","unstructured":"Snoek C., Worring M.: Multimodal video indexing: a review of the state-of-the-art. Multimedia Tools Appl. 25(1), 5\u201335 (2005)","journal-title":"Multimedia Tools Appl."},{"issue":"8","key":"217_CR7","doi-asserted-by":"crossref","first-page":"1026","DOI":"10.1109\/TPAMI.2002.1023800","volume":"24","author":"C. Carson","year":"2002","unstructured":"Carson C., Belongie S., Greenspan H., Malik J.: Blobworld: image segmentation using expectation-maximization and its application to image querying. IEEE Trans. Pattern Anal. Mach. Intell. 24(8), 1026\u20131038 (2002)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"2","key":"217_CR8","doi-asserted-by":"crossref","first-page":"189","DOI":"10.1109\/TMM.2002.1017733","volume":"4","author":"R. Zhao","year":"2002","unstructured":"Zhao R., Grosky W.I.: Narrowing the semantic gap: improved text-based web document retrieval using visual features. IEEE Trans. Multimedia 4(2), 189\u2013200 (2002)","journal-title":"IEEE Trans. Multimedia"},{"key":"217_CR9","doi-asserted-by":"crossref","unstructured":"Benitez, A., Chang, S.-F.: Semantic knowledge construction from annotated image collections. In: IEEE International Conference on Multimedia and Expo, vol. 2, pp. 205\u2013208 (2002)","DOI":"10.1109\/ICME.2002.1035549"},{"key":"217_CR10","doi-asserted-by":"crossref","unstructured":"Chang, S.-F., Manmatha, R., Chua, T.-S.: Combining Text and audio-visual features in video indexing. In: IEEE International Conference on Acoustics, Speech, and Signal Processing, vol. 5, Philadelphia, PA, March, pp. 1005\u20131008 (2005)","DOI":"10.1109\/ICASSP.2005.1416476"},{"issue":"2","key":"217_CR11","doi-asserted-by":"crossref","first-page":"137","DOI":"10.1023\/B:VISI.0000013087.49260.fb","volume":"57","author":"P. Viola","year":"2004","unstructured":"Viola P., Jones M.J.: Robust real-time face detection. Int. J. Comput. Vis. 57(2), 137\u2013154 (2004)","journal-title":"Int. J. Comput. Vis."},{"key":"217_CR12","doi-asserted-by":"crossref","unstructured":"Leibe, B., Seemann, E., Schiele, B.: Pedestrian detection in crowded scenes. In: IEEE Computer Society Conference on Computer Vision and Pattern Recognition, CVPR 2005, vol. 1, pp. 878\u2013885 (2005)","DOI":"10.1109\/CVPR.2005.272"},{"key":"217_CR13","first-page":"II-264","volume":"2","author":"R. Fergus","year":"2003","unstructured":"Fergus R., Perona P., Zisserman A.: Object class recognition by unsupervised scale-invariant learning. IEEE Comput. Soc. Conf. Comput. Vis. Pattern Recognit. 2, II-264\u2013II-271 (2003)","journal-title":"IEEE Comput. Soc. Conf. Comput. Vis. Pattern Recognit."},{"issue":"2","key":"217_CR14","doi-asserted-by":"crossref","first-page":"286","DOI":"10.1109\/TMM.2008.2009692","volume":"11","author":"L. Wu","year":"2003","unstructured":"Wu L., Hu Y., Li M., Yu N., Hua X.-S.: Scale-invariant visual language modeling for object categorization. IEEE Trans. Multimedia 11(2), 286\u2013294 (2003)","journal-title":"IEEE Trans. Multimedia"},{"issue":"4","key":"217_CR15","doi-asserted-by":"crossref","first-page":"594","DOI":"10.1109\/TPAMI.2006.79","volume":"28","author":"L. Fei-Fei","year":"2006","unstructured":"Fei-Fei L., Fergus R., Perona P.: One-shot learning of object categories. IEEE Trans. Pattern Anal. Mach. Intell. 28(4), 594\u2013611 (2006)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"9","key":"217_CR16","doi-asserted-by":"crossref","first-page":"1575","DOI":"10.1109\/TPAMI.2007.1155","volume":"29","author":"P. Quelhas","year":"2007","unstructured":"Quelhas P., Monay F., Odobez J.-M., Gatica-Perez D., Tuytelaars T.: A thousand words in a scene. IEEE Trans. Pattern Anal. Mach. Intell. 29(9), 1575\u20131589 (2007)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"217_CR17","unstructured":"Fei-Fei, L., Fergus, R., Perona, P.: Learning generative visual models from few training examples: an incremental bayesian approach tested on 101 object categories. In: IEEE CVPR 2004, Workshop on Generative-Model Based Vision (2004)"},{"key":"217_CR18","unstructured":"Caltech 101 Dataset Homepage [Online]. Available: http:\/\/www.vision.caltech.edu\/Image_Datasets\/Caltech101"},{"key":"217_CR19","unstructured":"Griffin, G., Holub, A., Perona, P.: Caltech-256 Object Category Dataset, California Institute of Technology, Tech. Rep. 7694 (2007)"},{"key":"217_CR20","doi-asserted-by":"crossref","unstructured":"Everingham, M., Zisserman, A., Williams, C., Gool, L. V., Allan, M., Bishop, C., Chapelle, O., Dalal, N., Deselaers, T., Dorko, G., Duffner, S., Eichhorn, J., Farquhar, J., Fritz, M., Garcia, C., Griffiths, T., Jurie, F., Keysers, D., Koskela, M., Laaksonen, J., Larlus, D., Leibe, B., Meng, H., Ney, H., Schiele, B., Schmid, C., Seemann, E., Shawe-Taylor, J., Storkey, A., Szedmak, S., Triggs, B., Ulusoy, I., Viitaniemi, V., Zhang, J.: The 2005 PASCAL Visual Object Classes Challenge. In: Selected Proceedings of the First PASCAL Challenges Workshop, LNAI, Springer-Verlag (2006)","DOI":"10.1007\/11736790_8"},{"key":"217_CR21","unstructured":"The PASCAL Visual Object Classes Homepage [Online]. Available: http:\/\/pascallin.ecs.soton.ac.uk\/challenges\/VOC"},{"issue":"1\u20133","key":"217_CR22","doi-asserted-by":"crossref","first-page":"157","DOI":"10.1007\/s11263-007-0090-8","volume":"77","author":"B.C. Russell","year":"2008","unstructured":"Russell B.C., Torralba A., Murphy K.P., Freeman W.T.: LabelMe: a database and web-based tool for image annotation. Int. J. Comput. Vis. 77(1\u20133), 157\u2013173 (2008)","journal-title":"Int. J. Comput. Vis."},{"key":"217_CR23","unstructured":"LabelMe Homepage [Online]. Available: http:\/\/labelme.csail.mit.edu"},{"issue":"10","key":"217_CR24","doi-asserted-by":"crossref","first-page":"1802","DOI":"10.1109\/TPAMI.2007.1097","volume":"29","author":"F. Monay","year":"2007","unstructured":"Monay F., Gatica-Perez D.: Modeling semantic aspects for cross-media image retrieval. Pattern Anal. Mach. Intell. 29(10), 1802\u20131817 (2007)","journal-title":"Pattern Anal. Mach. Intell."},{"key":"217_CR25","unstructured":"Getty Images [Online]. Available: http:\/\/www.gettyimages.com"},{"key":"217_CR26","unstructured":"Flickr Photo Sharing Service [Online]. Available: http:\/\/www.espgame.org\/gwap"},{"key":"217_CR27","doi-asserted-by":"crossref","unstructured":"von Ahn, L., Dabbish, L.: Labeling images with a computer game. In: ACM Conference on Human Factors in Computing Systems (CHI 2004), pp. 319\u2013326 (2004)","DOI":"10.1145\/985692.985733"},{"key":"217_CR28","unstructured":"The ESP Game [Online]. Available: http:\/\/www.flickr.com"},{"key":"217_CR29","unstructured":"Yahoo! News [Online]. Available: http:\/\/news.yahoo.com"},{"key":"217_CR30","doi-asserted-by":"crossref","unstructured":"Kender, J. R., Naphade, M. R.: Visual concepts for news story tracking: analyzing and exploiting the NIST TRECVID video annotation experiment. In: IEEE Computer Society Conference on Computer Vision and Pattern Recognition, vol. 1, pp. 1174\u20131181 (2005)","DOI":"10.1109\/CVPR.2005.371"},{"issue":"4","key":"217_CR31","doi-asserted-by":"crossref","first-page":"465","DOI":"10.1108\/00220410710758977","volume":"63","author":"P. Enser","year":"2007","unstructured":"Enser P., Sandom C.J., Hare J., Lewis P.: Facing the reality of semantic image retrieval. J. Document. 63(4), 465\u2013481 (2007)","journal-title":"J. Document."},{"key":"217_CR32","unstructured":"Maron, O., Ratan, A.L.: Multiple-instance learning for natural scene classification. In: The 15th International Conference on Machine Learning, pp. 341\u2013349 (1998)"},{"key":"217_CR33","doi-asserted-by":"crossref","unstructured":"Argillander, J., Iyengar, G., Nock, H.: Semantic annotation of multimedia using maximum entropy models. In: IEEE International Conference on Acoustics, Speech, and Signal Processing, vol. 2, Philadelphia, PA, USA, March 18\u201323, pp. 153\u2013156 (2005)","DOI":"10.1109\/ICASSP.2005.1415364"},{"key":"217_CR34","doi-asserted-by":"crossref","unstructured":"Carneiro, G., Vasconcelos, N.: Formulating semantic image annotation as a supervised learning problem. In: IEEE Conference on Computer Vision and Pattern Recognition, vol. 2, San Diego, June, pp. 163\u2013168 (2005)","DOI":"10.1109\/CVPR.2005.164"},{"issue":"9","key":"217_CR35","doi-asserted-by":"crossref","first-page":"1075","DOI":"10.1109\/TPAMI.2003.1227984","volume":"25","author":"J. Li","year":"2003","unstructured":"Li J., Wang J.: Automatic linguistic indexing of pictures by a statistical modeling approach. IEEE Trans. Pattern Anal. Mach. Intell. 25(9), 1075\u20131088 (2003)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"217_CR36","doi-asserted-by":"crossref","unstructured":"Jeon, J., Lavrenko, V., Manmatha, R.: Automatic image annotation and retrieval using cross-media relevance models. In: 26th Annual Int. ACM SIGIR Conference, Toronto, Canada, July 28\u2013August 1, pp. 119\u2013126 (2003)","DOI":"10.1145\/860435.860459"},{"key":"217_CR37","unstructured":"Lavrenko, V., Manmatha, R., Jeon, J.: A model for learning the semantics of pictures. In: 17th Annual Conference on Neural Information Processing Systems, vol. 16, pp. 553\u2013560 (2003)"},{"key":"217_CR38","doi-asserted-by":"crossref","unstructured":"Feng, S., Manmatha, R., Lavrenko, V.: Multiple Bernoulli relevance models for image and video annotation. In: International Conference on Computer Vision and Pattern Recognition, vol. 2, pp. 1002\u20131009 (2004)","DOI":"10.1109\/CVPR.2004.1315274"},{"key":"217_CR39","first-page":"1107","volume":"3","author":"K. Barnard","year":"2003","unstructured":"Barnard K., Duygulu P., de Freitas N., Forsyth D.A., Blei D., Jordan M.: Matching words and pictures. J. Mach. Learn. Res. 3, 1107\u20131135 (2003)","journal-title":"J. Mach. Learn. Res."},{"key":"217_CR40","doi-asserted-by":"crossref","unstructured":"Blei D., Jordan, M.I.: Modeling annotated data. In: 26th Annual International ACM SIGIR Conference, Toronto, Canada, July 28\u2013August 1, pp. 127\u2013134 (2003)","DOI":"10.1145\/860458.860460"},{"key":"217_CR41","doi-asserted-by":"crossref","unstructured":"Barnard, K., Forsyth, D.A.: Learning the semantics of words and pictures. In: International Conference on Computer Vision, vol. 2, pp. 408\u2013415 (2001)","DOI":"10.1109\/ICCV.2001.937654"},{"key":"217_CR42","unstructured":"Hofmann, T., Puzicha, J.: Statistical Models for Co-occurrence Data, AI Memo 1625, CBCL Memo 159, Artificial Intelligence Laboratory and Center for Biological and Computational Learning, MIT, Tech. Rep., February (1998)"},{"key":"217_CR43","doi-asserted-by":"crossref","unstructured":"Monay, F., Gatica-Perez, D.: PLSA-based image auto-annotation: constraining the latent space. In: ACM International Conference on Multimedia, October, pp. 348\u2013351 (2004)","DOI":"10.1145\/1027527.1027608"},{"key":"217_CR44","unstructured":"Mori, Y., Takahashi, H., Oka, R.: Image-to-word transformation based on dividing and vector quantizing images with words. In: 1st International Workshop on Multimedia Intelligent Storage and Retrieval Management (1999)"},{"key":"217_CR45","doi-asserted-by":"crossref","unstructured":"Duygulu, P., Barnard, K., Freitas, N., Forsyth, D.A.: Object recognition as machine translation: learning a lexicon for a fixed image vocabulary. In: 7th European Conference on Computer Vision, vol. 4, Copenhagen Denmark, May 27\u2013June 2, pp. 97\u2013112 (2002)","DOI":"10.1007\/3-540-47979-1_7"},{"key":"217_CR46","unstructured":"Pan, J.-Y., Yang, H.-J., Duygulu, P., Faloutsos, C.: Automatic image captioning. In: The 2004 IEEE International Conference on Multimedia and Expo, vol. 3, Taipei, Taiwan, June, pp. 1987\u20131990 (2004)"},{"key":"217_CR47","doi-asserted-by":"crossref","unstructured":"Carbonetto, P., de Freitas, N., Barnard, K.: A statistical model for general contextual object recognition. In: 8th European Conference on Computer Vision, Prague, Czech Republic, May 11\u201314, pp. 350\u2013362 (2004)","DOI":"10.1007\/978-3-540-24670-1_27"},{"issue":"2","key":"217_CR48","first-page":"263","volume":"19","author":"P. Brown","year":"1993","unstructured":"Brown P., Pietra S.A.D., Pietra V.J.D., Mercer R.L.: The mathematics of statistical machine translation: parameter estimation. Comput. Linguist. 19(2), 263\u2013311 (1993)","journal-title":"Comput. Linguist."},{"issue":"8","key":"217_CR49","doi-asserted-by":"crossref","first-page":"888","DOI":"10.1109\/34.868688","volume":"22","author":"J. Shi","year":"2000","unstructured":"Shi J., Malik J.: Normalized cuts and image segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 22(8), 888\u2013905 (2000)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"217_CR50","doi-asserted-by":"crossref","unstructured":"Barnard, K., Duygulu, P., Guru, R., Gabbur, P., Forsyth, D.: The effects of segmentation and feature choice in a translation model of object recognition. In: IEEE Computer Society Conference on Computer Vision and Pattern Recognition, vol. 2, Madison, Wisconsin, June, pp. 675\u2013682 (2003)","DOI":"10.1109\/CVPR.2003.1211532"},{"key":"217_CR51","doi-asserted-by":"crossref","unstructured":"Jeon, J., Manmatha, R.: Using maximum entropy for automatic image annotation. In: 3rd International Conference on Image and Video Retrieval, Ireland, July 21\u201323, pp. 24\u201332 (2004)","DOI":"10.1007\/978-3-540-27814-6_7"},{"key":"217_CR52","doi-asserted-by":"crossref","unstructured":"Virga P., Duygulu, P.: Systematic evaluation of machine translation methods for image and video annotation. In: The 4th International Conference on Image and Video Retrieval, Singapore, July 20\u201322, pp. 174\u2013183 (2005)","DOI":"10.1007\/11526346_21"},{"issue":"10","key":"217_CR53","doi-asserted-by":"crossref","first-page":"1802","DOI":"10.1109\/TPAMI.2007.1097","volume":"29","author":"F. Monay","year":"2007","unstructured":"Monay F., Gatica-Perez D.: Modeling semantic aspects for cross-media image indexing. IEEE Trans. Pattern Anal. Mach. Intell. 29(10), 1802\u20131817 (2007)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"217_CR54","unstructured":"TREC Video Retrieval Evaluation [Online]. Available: http:\/\/www-nlpir.nist.gov\/projects\/trecvid"},{"key":"217_CR55","doi-asserted-by":"crossref","unstructured":"Smeaton, A.F., Over, P., Kraaij, W.: Evaluation campaigns and TRECVid, In: 8th ACM International Workshop on Multimedia Information Retrieval, pp. 321\u2013330 (2006)","DOI":"10.1145\/1178677.1178722"},{"key":"217_CR56","doi-asserted-by":"crossref","unstructured":"Smeaton, A., Over, P., Kraaij, W.: High level feature detection from video in TRECVid: a 5-year retrospective of achievements. In: Divakaran, A. (ed.) Multimedia Content Analysis, Theory and Applications. Springer, Berlin (2008)","DOI":"10.1007\/978-0-387-76569-3_6"},{"key":"217_CR57","doi-asserted-by":"crossref","unstructured":"Ghoshal, A., Ircing, P., Khudanpur, S.: Hidden Markov models for automatic annotation and content based retrieval of images and video. In: The 28th International ACM SIGIR Conference, Salvador, Brazil, August 15\u201319, pp. 544\u2013551 (2005)","DOI":"10.1145\/1076034.1076127"},{"key":"217_CR58","unstructured":"Duygulu, P., Hauptmann, A.: What\u2019s news what\u2019s not? Associating News videos with words. In: The 3rd International Conference on Image and Video Retrieval (CIVR 2004), Ireland, July 21\u201323, pp. 21\u201323 (2004)"},{"key":"217_CR59","unstructured":"Wactlar, H., Hauptmann, A., Witbrock, M.: Informedia News-On Demand: Using Speech Recognition to Create a Digital Video Library, CMU Technical Report, CMU-CS-98-109, Tech. Rep. (1998)"},{"key":"217_CR60","doi-asserted-by":"crossref","unstructured":"Gross, R., Baker, S., Matthews, I., Kanade, T.: Face recognition across pose and illumination. In: Li, S.Z., Jain, A.K. (eds.) Handbook of Face Recognition. Springer, Berlin, pp. 193\u2013216 (2004)","DOI":"10.1007\/0-387-27257-7_10"},{"issue":"4","key":"217_CR61","doi-asserted-by":"crossref","first-page":"399","DOI":"10.1145\/954339.954342","volume":"35","author":"W. Zhao","year":"2003","unstructured":"Zhao W., Chellappa R., Phillips P., Rosenfeld A.: Face recognition: a literature survey. ACM Comput. Surv. 35(4), 399\u2013458 (2003)","journal-title":"ACM Comput. Surv."},{"key":"217_CR62","doi-asserted-by":"crossref","unstructured":"Yang, J., Chen, M.-Y., Hauptmann, A.: Finding Person X: correlating names with visual appearances. In: International Conference on Image and Video Retrieval, Ireland, pp. 270\u2013278 (2004)","DOI":"10.1007\/978-3-540-27814-6_34"},{"key":"217_CR63","unstructured":"Berg, T., Berg, A.C., Edwards, J., Maire, M., White, R., Teh, Y.-W., Learned-Miller, E., Forsyth, D.: Faces and names in the news. In: IEEE Conference on Computer Vision and Pattern Recognition (2004)"},{"key":"217_CR64","doi-asserted-by":"crossref","unstructured":"Ozkan, D., Duygulu, P.: A graph based approach for naming faces in news photos. In: IEEE International Conference on Computer Vision and Pattern Recognition, vol. 2, pp. 1477\u20131482 (2006)","DOI":"10.1109\/CVPR.2006.29"},{"key":"217_CR65","unstructured":"Satoh, S., Kanade, T.: Name-It: Association of face and name in video. In: IEEE Conference on Computer Vision and Pattern Recognition (1997)"},{"key":"217_CR66","unstructured":"Ba\u015ftan, M., Duygulu, P.: Recognizing objects and scenes in news videos. In: The International Conference on Image and Video Retrieval, Lecture Notes in Computer Science, 40071, pp. 380\u2013390 (2006)"},{"issue":"5","key":"217_CR67","doi-asserted-by":"crossref","first-page":"923","DOI":"10.1109\/TMM.2007.900138","volume":"9","author":"N. Rasiwasia","year":"2007","unstructured":"Rasiwasia N., Vasconcelos N.: Bridging the semantic gap: query by semantic example. IEEE Trans. Multimedia 9(5), 923\u2013938 (2007)","journal-title":"IEEE Trans. Multimedia"},{"key":"217_CR68","unstructured":"Giza++: Training of statistical translation models [Online]. Available: http:\/\/www.fjoch.com\/GIZA++.html"},{"issue":"29","key":"217_CR69","doi-asserted-by":"crossref","first-page":"19","DOI":"10.1162\/089120103321337421","volume":"1","author":"F.J. Och","year":"2003","unstructured":"Och F.J., Ney H.: A systematic comparison of various statistical alignment models. Comput. Linguist. 1(29), 19\u201351 (2003)","journal-title":"Comput. Linguist."},{"key":"217_CR70","unstructured":"Lin, C. -Y., Tseng, B. L., Smith, J. R.: Video Collaborative annotation forum: establishing ground-truth labels on large multimedia datasets. In: NIST TREC-2003 Video Retrieval Evaluation Conference, Gaithersburg, MD, November (2003)"},{"issue":"3","key":"217_CR71","doi-asserted-by":"crossref","first-page":"86","DOI":"10.1109\/MMUL.2006.63","volume":"13","author":"M. Naphade","year":"2006","unstructured":"Naphade M., Curtis J., Hauptmann A., Kennedy L., Hsu W., Chang S.-F., Smith J.: Large-scale concept ontology for multimedia. IEEE Multimedia 13(3), 86\u201391 (2006)","journal-title":"IEEE Multimedia"},{"issue":"1-2","key":"217_CR72","doi-asserted-by":"crossref","first-page":"89","DOI":"10.1016\/S0167-6393(01)00061-9","volume":"37","author":"J. Gauvain","year":"2002","unstructured":"Gauvain J., Lamel L., Adda G.: The LIMSI broadcast news transcription system. Speech Commun. 37(1-2), 89\u2013108 (2002)","journal-title":"Speech Commun."},{"issue":"2","key":"217_CR73","doi-asserted-by":"crossref","first-page":"91","DOI":"10.1023\/B:VISI.0000029664.99615.94","volume":"60","author":"D.G. Lowe","year":"2004","unstructured":"Lowe D.G.: Distinctive image features from scale-invariant keypoints. Int. J. Comput. Vis. 60(2), 91\u2013110 (2004)","journal-title":"Int. J. Comput. Vis."},{"key":"217_CR74","doi-asserted-by":"crossref","unstructured":"Sivic, J., Zisserman, A.: Video google: a text retrieval approach to object matching in videos. In: Proceedings of the Ninth IEEE International Conference on Computer Vision (ICCV) (2003)","DOI":"10.1109\/ICCV.2003.1238663"},{"key":"217_CR75","doi-asserted-by":"crossref","first-page":"235","DOI":"10.1093\/ijl\/3.4.235","volume":"3","author":"G.A. Miller","year":"1990","unstructured":"Miller G.A., Beckwith R., Fellbaum C., Gross D., Miller K.J.: Introduction to WordNet: an online lexical database. Int. J. Lexicogr. 3, 235\u2013244 (1990)","journal-title":"Int. J. Lexicogr."},{"issue":"3","key":"217_CR76","doi-asserted-by":"crossref","first-page":"384","DOI":"10.1109\/TCSVT.2006.888941","volume":"17","author":"J. Tang","year":"2007","unstructured":"Tang J., Lewis P.: A study of quality issues for image auto-annotation with the corel data-set. IEEE Trans. Circuits Syst. Video Technol. 17(3), 384\u2013389 (2007)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"217_CR77","unstructured":"Joachims, T.: Multi-class support vector machine [Online]. Available: http:\/\/svmlight.joachims.org\/svm-multiclass.html"}],"container-title":["Machine Vision and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-009-0217-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s00138-009-0217-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-009-0217-8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,12]],"date-time":"2025-02-12T12:12:32Z","timestamp":1739362352000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s00138-009-0217-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2009,9,30]]},"references-count":77,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2011,1]]}},"alternative-id":["217"],"URL":"https:\/\/doi.org\/10.1007\/s00138-009-0217-8","relation":{},"ISSN":["0932-8092","1432-1769"],"issn-type":[{"type":"print","value":"0932-8092"},{"type":"electronic","value":"1432-1769"}],"subject":[],"published":{"date-parts":[[2009,9,30]]}}}