{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T07:02:35Z","timestamp":1785308555264,"version":"3.55.0"},"publisher-location":"Cham","reference-count":51,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783319106014","type":"print"},{"value":"9783319106021","type":"electronic"}],"license":[{"start":{"date-parts":[[2014,1,1]],"date-time":"2014-01-01T00:00:00Z","timestamp":1388534400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2014]]},"DOI":"10.1007\/978-3-319-10602-1_48","type":"book-chapter","created":{"date-parts":[[2014,8,14]],"date-time":"2014-08-14T03:07:56Z","timestamp":1407985676000},"page":"740-755","source":"Crossref","is-referenced-by-count":25992,"title":["Microsoft COCO: Common Objects in Context"],"prefix":"10.1007","author":[{"given":"Tsung-Yi","family":"Lin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Michael","family":"Maire","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Serge","family":"Belongie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"James","family":"Hays","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pietro","family":"Perona","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Deva","family":"Ramanan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Piotr","family":"Doll\u00e1r","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"C. Lawrence","family":"Zitnick","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","reference":[{"key":"48_CR1","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: ImageNet: A Large-Scale Hierarchical Image Database. In: CVPR (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"48_CR2","doi-asserted-by":"crossref","unstructured":"Everingham, M., Van Gool, L., Williams, C.K.I., Winn, J., Zisserman, A.: The PASCAL visual object classes (VOC) challenge. IJCV 88(2), 303\u2013338 (2010)","DOI":"10.1007\/s11263-009-0275-4"},{"key":"48_CR3","doi-asserted-by":"crossref","unstructured":"Xiao, J., Hays, J., Ehinger, K.A., Oliva, A., Torralba, A.: SUN database: Large-scale scene recognition from abbey to zoo. In: CVPR (2010)","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"48_CR4","doi-asserted-by":"crossref","unstructured":"Doll\u00e1r, P., Wojek, C., Schiele, B., Perona, P.: Pedestrian detection: An evaluation of the state of the art. PAMI 34 (2012)","DOI":"10.1109\/TPAMI.2011.155"},{"key":"48_CR5","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.: ImageNet classification with deep convolutional neural networks. In: NIPS (2012)"},{"key":"48_CR6","doi-asserted-by":"crossref","unstructured":"Girshick, R., Donahue, J., Darrell, T., Malik, J.: Rich feature hierarchies for accurate object detection and semantic segmentation. In: CVPR (2014)","DOI":"10.1109\/CVPR.2014.81"},{"key":"48_CR7","unstructured":"Sermanet, P., Eigen, D., Zhang, S., Mathieu, M., Fergus, R., LeCun, Y.: OverFeat: Integrated recognition, localization and detection using convolutional networks. In: ICLR (April 2014)"},{"key":"48_CR8","doi-asserted-by":"crossref","unstructured":"Farhadi, A., Endres, I., Hoiem, D., Forsyth, D.: Describing objects by their attributes. In: CVPR (2009)","DOI":"10.1109\/CVPR.2009.5206772"},{"key":"48_CR9","doi-asserted-by":"crossref","unstructured":"Patterson, G., Hays, J.: SUN attribute database: Discovering, annotating, and recognizing scene attributes. In: CVPR (2012)","DOI":"10.1109\/CVPR.2012.6247998"},{"key":"48_CR10","doi-asserted-by":"crossref","unstructured":"Bourdev, L., Malik, J.: Poselets: Body part detectors trained using 3D human pose annotations. In: ICCV (2009)","DOI":"10.1109\/ICCV.2009.5459303"},{"key":"48_CR11","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"746","DOI":"10.1007\/978-3-642-33715-4_54","volume-title":"Computer Vision \u2013 ECCV 2012","author":"N. Silberman","year":"2012","unstructured":"Silberman, N., Hoiem, D., Kohli, P., Fergus, R.: Indoor segmentation and support inference from RGBD images. In: Fitzgibbon, A., Lazebnik, S., Perona, P., Sato, Y., Schmid, C. (eds.) ECCV 2012, Part V. LNCS, vol.\u00a07576, pp. 746\u2013760. Springer, Heidelberg (2012)"},{"key":"48_CR12","unstructured":"Palmer, S., Rosch, E., Chase, P.: Canonical perspective and the perception of objects. Attention and Performance IX 1,\u00a04 (1981)"},{"key":"48_CR13","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"340","DOI":"10.1007\/978-3-642-33712-3_25","volume-title":"Computer Vision \u2013 ECCV 2012","author":"D. Hoiem","year":"2012","unstructured":"Hoiem, D., Chodpathumwan, Y., Dai, Q.: Diagnosing error in object detectors. In: Fitzgibbon, A., Lazebnik, S., Perona, P., Sato, Y., Schmid, C. (eds.) ECCV 2012, Part III. LNCS, vol.\u00a07574, pp. 340\u2013353. Springer, Heidelberg (2012)"},{"issue":"2","key":"48_CR14","doi-asserted-by":"publisher","first-page":"88","DOI":"10.1016\/j.patrec.2008.04.005","volume":"30","author":"G. Brostow","year":"2009","unstructured":"Brostow, G., Fauqueur, J., Cipolla, R.: Semantic object classes in video: A high-definition ground truth database. PRL\u00a030(2), 88\u201397 (2009)","journal-title":"PRL"},{"issue":"1-3","key":"48_CR15","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1007\/s11263-007-0090-8","volume":"77","author":"B. Russell","year":"2008","unstructured":"Russell, B., Torralba, A., Murphy, K., Freeman, W.: LabelMe: a database and web-based tool for image annotation. IJCV\u00a077(1-3), 157\u2013173 (2008)","journal-title":"IJCV"},{"key":"48_CR16","doi-asserted-by":"crossref","unstructured":"Bell, S., Upchurch, P., Snavely, N., Bala, K.: OpenSurfaces: A richly annotated catalog of surface appearance. SIGGRAPH 32(4) (2013)","DOI":"10.1145\/2461912.2462002"},{"key":"48_CR17","unstructured":"Ordonez, V., Kulkarni, G., Berg, T.: Im2text: Describing images using 1 million captioned photographs. In: NIPS (2011)"},{"key":"48_CR18","doi-asserted-by":"crossref","unstructured":"Deng, J., Russakovsky, O., Krause, J., Bernstein, M., Berg, A., Fei-Fei, L.: Scalable multi-label annotation. In: CHI (2014)","DOI":"10.1145\/2556288.2557011"},{"key":"48_CR19","doi-asserted-by":"crossref","unstructured":"Lin, T., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft COCO: Common objects in context. CoRR abs\/1405.0312 (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"48_CR20","unstructured":"Scharstein, D., Szeliski, R.: A taxonomy and evaluation of dense two-frame stereo correspondence algorithms. IJCV 47(1-3), 7\u201342 (2002)"},{"issue":"1","key":"48_CR21","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11263-010-0390-2","volume":"92","author":"S. Baker","year":"2011","unstructured":"Baker, S., Scharstein, D., Lewis, J., Roth, S., Black, M., Szeliski, R.: A database and evaluation methodology for optical flow. IJCV\u00a092(1), 1\u201331 (2011)","journal-title":"IJCV"},{"key":"48_CR22","unstructured":"Fei-Fei, L., Fergus, R., Perona, P.: Learning generative visual models from few training examples: An incremental bayesian approach tested on 101 object categories. In: CVPR Workshop of Generative Model Based Vision, WGMBV (2004)"},{"key":"48_CR23","unstructured":"Griffin, G., Holub, A., Perona, P.: Caltech-256 object category dataset. Technical Report 7694, California Institute of Technology (2007)"},{"key":"48_CR24","unstructured":"Dalal, N., Triggs, B.: Histograms of oriented gradients for human detection. In: CVPR (2005)"},{"key":"48_CR25","unstructured":"Lecun, Y., Cortes, C.: The MNIST database of handwritten digits (1998)"},{"key":"48_CR26","unstructured":"Nene, S.A., Nayar, S.K., Murase, H.: Columbia object image library (coil-20). Technical report, Columbia Universty (1996)"},{"key":"48_CR27","unstructured":"Krizhevsky, A., Hinton, G.: Learning multiple layers of features from tiny images. Computer Science Department, University of Toronto, Tech. Rep. (2009)"},{"issue":"11","key":"48_CR28","doi-asserted-by":"publisher","first-page":"1958","DOI":"10.1109\/TPAMI.2008.128","volume":"30","author":"A. Torralba","year":"2008","unstructured":"Torralba, A., Fergus, R., Freeman, W.T.: 80 million tiny images: A large data set for nonparametric object and scene recognition. PAMI\u00a030(11), 1958\u20131970 (2008)","journal-title":"PAMI"},{"key":"48_CR29","doi-asserted-by":"crossref","unstructured":"Ordonez, V., Deng, J., Choi, Y., Berg, A., Berg, T.: From large scale image categorization to entry-level categories. In: ICCV (2013)","DOI":"10.1109\/ICCV.2013.344"},{"key":"48_CR30","doi-asserted-by":"crossref","unstructured":"Fellbaum, C.: WordNet: An electronic lexical database. Blackwell Books (1998)","DOI":"10.7551\/mitpress\/7287.001.0001"},{"key":"48_CR31","unstructured":"Welinder, P., Branson, S., Mita, T., Wah, C., Schroff, F., Belongie, S., Perona, P.: Caltech-UCSD Birds 200. Technical Report CNS-TR-201, Caltech. (2010)"},{"key":"48_CR32","doi-asserted-by":"crossref","unstructured":"Hjelm\u00e5s, E., Low, B.: Face detection: A survey. CVIU 83(3), 236\u2013274 (2001)","DOI":"10.1006\/cviu.2001.0921"},{"key":"48_CR33","unstructured":"Huang, G.B., Ramesh, M., Berg, T., Learned-Miller, E.: Labeled faces in the wild. Technical Report 07-49, University of Massachusetts, Amherst (October 2007)"},{"key":"48_CR34","doi-asserted-by":"crossref","unstructured":"Russakovsky, O., Deng, J., Huang, Z., Berg, A., Fei-Fei, L.: Detecting avocados to zucchinis: what have we done, and where are we going? In: ICCV (2013)","DOI":"10.1109\/ICCV.2013.258"},{"issue":"1","key":"48_CR35","doi-asserted-by":"publisher","first-page":"2","DOI":"10.1007\/s11263-007-0109-1","volume":"81","author":"J. Shotton","year":"2009","unstructured":"Shotton, J., Winn, J., Rother, C., Criminisi, A.: TextonBoost for image understanding: Multi-class object recognition and segmentation by jointly modeling texture, layout, and context. IJCV\u00a081(1), 2\u201323 (2009)","journal-title":"IJCV"},{"key":"48_CR36","unstructured":"Seitz, S.M., Curless, B., Diebel, J., Scharstein, D., Szeliski, R.: A comparison and evaluation of multi-view stereo reconstruction algorithms. In: CVPR (2006)"},{"issue":"5","key":"48_CR37","doi-asserted-by":"publisher","first-page":"898","DOI":"10.1109\/TPAMI.2010.161","volume":"33","author":"P. Arbelaez","year":"2011","unstructured":"Arbelaez, P., Maire, M., Fowlkes, C., Malik, J.: Contour detection and hierarchical image segmentation. PAMI\u00a033(5), 898\u2013916 (2011)","journal-title":"PAMI"},{"key":"48_CR38","doi-asserted-by":"crossref","unstructured":"Lampert, C., Nickisch, H., Harmeling, S.: Learning to detect unseen object classes by between-class attribute transfer. In: CVPR (2009)","DOI":"10.1109\/CVPRW.2009.5206594"},{"key":"48_CR39","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1007\/978-3-540-88682-2_4","volume-title":"Computer Vision \u2013 ECCV 2008","author":"G. Heitz","year":"2008","unstructured":"Heitz, G., Koller, D.: Learning spatial context: Using stuff to find things. In: Forsyth, D., Torr, P., Zisserman, A. (eds.) ECCV 2008, Part I. LNCS, vol.\u00a05302, pp. 30\u201343. Springer, Heidelberg (2008)"},{"key":"48_CR40","unstructured":"Sitton, R.: Spelling Sourcebook. Egger Publishing (1996)"},{"key":"48_CR41","doi-asserted-by":"crossref","unstructured":"Berg, T., Berg, A.: Finding iconic images. In: CVPR (2009)","DOI":"10.1109\/CVPR.2009.5204174"},{"key":"48_CR42","doi-asserted-by":"crossref","unstructured":"Torralba, A., Efros, A.: Unbiased look at dataset bias. In: CVPR (2011)","DOI":"10.1109\/CVPR.2011.5995347"},{"key":"48_CR43","doi-asserted-by":"crossref","unstructured":"Douze, M., J\u00e9gou, H., Sandhawalia, H., Amsaleg, L., Schmid, C.: Evaluation of gist descriptors for web-scale image search. In: CIVR (2009)","DOI":"10.1145\/1646396.1646421"},{"issue":"9","key":"48_CR44","doi-asserted-by":"publisher","first-page":"1627","DOI":"10.1109\/TPAMI.2009.167","volume":"32","author":"P. Felzenszwalb","year":"2010","unstructured":"Felzenszwalb, P., Girshick, R., McAllester, D., Ramanan, D.: Object detection with discriminatively trained part-based models. PAMI\u00a032(9), 1627\u20131645 (2010)","journal-title":"PAMI"},{"key":"48_CR45","unstructured":"Girshick, R., Felzenszwalb, P., McAllester, D.: Discriminatively trained deformable part models, release 5. PAMI (2012)"},{"key":"48_CR46","doi-asserted-by":"crossref","unstructured":"Zhu, X., Vondrick, C., Ramanan, D., Fowlkes, C.: Do we need more training data or better models for object detection? In: BMVC (2012)","DOI":"10.5244\/C.26.80"},{"key":"48_CR47","doi-asserted-by":"crossref","unstructured":"Brox, T., Bourdev, L., Maji, S., Malik, J.: Object segmentation by alignment of poselet activations to image contours. In: CVPR (2011)","DOI":"10.1109\/CVPR.2011.5995659"},{"issue":"9","key":"48_CR48","doi-asserted-by":"publisher","first-page":"1731","DOI":"10.1109\/TPAMI.2011.208","volume":"34","author":"Y. Yang","year":"2012","unstructured":"Yang, Y., Hallman, S., Ramanan, D., Fowlkes, C.: Layered object models for image segmentation. PAMI\u00a034(9), 1731\u20131743 (2012)","journal-title":"PAMI"},{"key":"48_CR49","doi-asserted-by":"crossref","unstructured":"Ramanan, D.: Using segmentation to verify object hypotheses. In: CVPR (2007)","DOI":"10.1109\/CVPR.2007.383271"},{"key":"48_CR50","unstructured":"Dai, Q., Hoiem, D.: Learning to localize detected objects. In: CVPR (2012)"},{"key":"48_CR51","unstructured":"Rashtchian, C., Young, P., Hodosh, M., Hockenmaier, J.: Collecting image annotations using Amazon\u2019s Mechanical Turk. In: NAACL Workshop (2010)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2014"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-10602-1_48","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,12,2]],"date-time":"2019-12-02T09:36:21Z","timestamp":1575279381000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-10602-1_48"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014]]},"ISBN":["9783319106014","9783319106021"],"references-count":51,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014]]}}}