{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T08:39:48Z","timestamp":1774600788518,"version":"3.50.1"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2014,11,4]],"date-time":"2014-11-04T00:00:00Z","timestamp":1415059200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2015,4]]},"DOI":"10.1007\/s11263-014-0780-y","type":"journal-article","created":{"date-parts":[[2014,11,3]],"date-time":"2014-11-03T06:11:01Z","timestamp":1414995061000},"page":"188-203","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":54,"title":["Towards Scene Understanding with Detailed 3D Object Representations"],"prefix":"10.1007","volume":"112","author":[{"given":"M. Zeeshan","family":"Zia","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Michael","family":"Stark","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Konrad","family":"Schindler","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,11,4]]},"reference":[{"key":"780_CR1","doi-asserted-by":"crossref","unstructured":"Bao, S. Y., & Savarese, S. (2011). Semantic structure from motion. In CVPR.","DOI":"10.1109\/CVPR.2011.5995462"},{"key":"780_CR2","unstructured":"Belongie, S., Malik, J., & Puzicha, J. (2000). Shape context: A new descriptor for shape matching and object recognition. In NIPS."},{"key":"780_CR3","doi-asserted-by":"crossref","unstructured":"Bourdev, L., & Malik, J. (2009). Poselets: Body part detectors trained using 3D human pose annotations. In ICCV.","DOI":"10.1109\/ICCV.2009.5459303"},{"key":"780_CR4","doi-asserted-by":"crossref","first-page":"285","DOI":"10.1016\/0004-3702(81)90028-X","volume":"17","author":"RA Brooks","year":"1981","unstructured":"Brooks, R. A. (1981). Symbolic reasoning among 3-d models and 2-d images. Artificial Intelligence, 17, 285\u2013348.","journal-title":"Artificial Intelligence"},{"key":"780_CR5","doi-asserted-by":"crossref","unstructured":"Choi, W., Chao, Y. -W., Pantofaru, C., & Savarese, S. (2013). Understanding indoor scenes using 3D geometric phrases. In CVPR.","DOI":"10.1109\/CVPR.2013.12"},{"issue":"1","key":"780_CR6","doi-asserted-by":"crossref","first-page":"38","DOI":"10.1006\/cviu.1995.1004","volume":"61","author":"TF Cootes","year":"1995","unstructured":"Cootes, T. F., Taylor, C. J., Cooper, D. H., & Graham, J. (1995). Active shape models, their training and application. Computer Vision and Image Understanding, 61(1), 38\u201359.","journal-title":"Computer Vision and Image Understanding"},{"key":"780_CR7","doi-asserted-by":"crossref","unstructured":"Dalal, N., Triggs, B. (2005). Histograms of oriented gradients for human detection. In CVPR.","DOI":"10.1109\/CVPR.2005.177"},{"key":"780_CR8","unstructured":"Del Pero, L., Bowdish, J., Kermgard, B., Hartley, E., & Barnard, K. (2013). Understanding Bayesian rooms using composite 3D object models. In CVPR."},{"key":"780_CR9","doi-asserted-by":"crossref","unstructured":"Enzweiler, M., Eigenstetter, A., Schiele, B., & Gavrila, D. M. (2010). Multi-Cue pedestrian classification with partial occlusion handling. In CVPR.","DOI":"10.1109\/CVPR.2010.5540111"},{"issue":"10","key":"780_CR10","doi-asserted-by":"crossref","first-page":"1831","DOI":"10.1109\/TPAMI.2009.109","volume":"31","author":"A Ess","year":"2009","unstructured":"Ess, A., Leibe, B., Schindler, K., & Gool, L. V. (2009). Robust multi-person tracking from a mobile platform. Pattern Analysis and Machine Intelligence, 31(10), 1831\u20131846.","journal-title":"Pattern Analysis and Machine Intelligence"},{"issue":"2","key":"780_CR11","doi-asserted-by":"crossref","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham, M., Van Gool, L., Williams, C. K., Winn, J., & Zisserman, A. (2010). The pascal visual object classes (VOC) challenge. International Journal of Computer Vision, 88(2), 303\u2013338.","journal-title":"International Journal of Computer Vision"},{"issue":"9","key":"780_CR12","doi-asserted-by":"crossref","first-page":"1627","DOI":"10.1109\/TPAMI.2009.167","volume":"32","author":"PF Felzenszwalb","year":"2010","unstructured":"Felzenszwalb, P. F., Girshick, R., McAllester, D., & Ramanan, D. (2010). Object detection with discriminatively trained part based models. Pattern Analysis and Machine Intelligence, 32(9), 1627\u20131645.","journal-title":"Pattern Analysis and Machine Intelligence"},{"key":"780_CR13","doi-asserted-by":"crossref","unstructured":"Fransens, R., Strecha, C., & Gool, L. V. (2006). A mean field EM-algorithm for coherent occlusion handling in MAP-estimation. In CVPR.","DOI":"10.1109\/CVPR.2006.31"},{"key":"780_CR14","doi-asserted-by":"crossref","unstructured":"Gao, T., Packer, B., & Koller, D. (2011). A segmentation-aware object detection model with occlusion handling. In CVPR.","DOI":"10.1109\/CVPR.2011.5995623"},{"key":"780_CR15","unstructured":"Geiger, A., Lenz, P., & Urtasun, R. (2012). Are we ready for autonomous driving?. The KITTI vision benchmark suite. In CVPR."},{"key":"780_CR16","unstructured":"Geiger, A., Wojek, C., & Urtasun, R. (2011). Joint 3D estimation of objects and scene layout. In NIPS."},{"key":"780_CR17","unstructured":"Girshick, R. B., Felzenszwalb, P. F., & McAllester, D. (2011). Object detection with grammar models. In NIPS."},{"key":"780_CR18","doi-asserted-by":"crossref","unstructured":"Gupta, A., Efros, A. A., & Hebert, M. (2010). Blocks world revisited: Image understanding using qualitative geometry and mechanics. In ECCV.","DOI":"10.1007\/978-3-642-15561-1_35"},{"issue":"3","key":"780_CR19","doi-asserted-by":"crossref","first-page":"295","DOI":"10.1023\/A:1008112528134","volume":"35","author":"M Haag","year":"1999","unstructured":"Haag, M., & Nagel, H.-H. (1999). Combination of edge element and optical flow estimates for 3d-model-based vehicle tracking in traffic image sequences. International Journal of Computer Vision, 35(3), 295\u2013319.","journal-title":"International Journal of Computer Vision"},{"key":"780_CR20","doi-asserted-by":"crossref","unstructured":"Hedau, V., Hoiem, D., & Forsyth, D. A. (2010). Thinking inside the box: Using appearance models and context based on room geometry. In ECCV.","DOI":"10.1007\/978-3-642-15567-3_17"},{"key":"780_CR21","unstructured":"Hejrati, M., & Ramanan, D. (2012). Analyzing 3D objects in cluttered images. In NIPS."},{"issue":"1","key":"780_CR22","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1007\/s11263-008-0137-5","volume":"80","author":"D Hoiem","year":"2008","unstructured":"Hoiem, D., Efros, A., & Hebert, M. (2008). Putting objects in perspective. International Journal of Computer Vision, 80(1), 3\u201315.","journal-title":"International Journal of Computer Vision"},{"key":"780_CR23","doi-asserted-by":"crossref","first-page":"279","DOI":"10.1016\/0004-3702(80)90004-1","volume":"13","author":"T Kanade","year":"1980","unstructured":"Kanade, T. (1980). A theory of Origami world. Artificial Intelligence, 13, 279\u2013311.","journal-title":"Artificial Intelligence"},{"issue":"3","key":"780_CR24","doi-asserted-by":"crossref","first-page":"257","DOI":"10.1007\/BF01539538","volume":"10","author":"D Koller","year":"1993","unstructured":"Koller, D., Daniilidis, K., & Nagel, H. H. (1993). Model-based object tracking in monocular image sequences of road traffic scenes. International Journal of Computer Vision, 10(3), 257\u2013281.","journal-title":"International Journal of Computer Vision"},{"key":"780_CR25","unstructured":"Kwak, S., Nam, W., Han, B., & Han, J. H. (2011). Learning occlusion with likelihoods for visual tracking. In ICCV."},{"key":"780_CR26","doi-asserted-by":"crossref","unstructured":"Leibe, B., Leonardis, A., & Schiele, B. (2006). An implicit shape model for combined object categorization and segmentation. In Toward category-level object recognition.","DOI":"10.1007\/11957959_26"},{"key":"780_CR27","doi-asserted-by":"crossref","unstructured":"Leordeanu, M., & Hebert, M. (2008). Smoothing-based optimization. In CVPR.","DOI":"10.1109\/CVPR.2008.4587482"},{"issue":"9","key":"780_CR28","doi-asserted-by":"crossref","first-page":"1860","DOI":"10.1109\/TPAMI.2011.40","volume":"33","author":"Y Li","year":"2011","unstructured":"Li, Y., Gu, L., & Kanade, T. (2011). Robustly aligning a shape model and its application to car alignment of unknown pose. Pattern Analysis and Machine Intelligence, 33(9), 1860\u20131876.","journal-title":"Pattern Analysis and Machine Intelligence"},{"key":"780_CR29","doi-asserted-by":"crossref","unstructured":"Liu, X., Zhao, Y., & Zhu, S. -C. (2014). Single-view 3D scene parsing by attributed grammar. In CVPR.","DOI":"10.1109\/CVPR.2014.93"},{"issue":"3","key":"780_CR30","doi-asserted-by":"crossref","first-page":"355","DOI":"10.1016\/0004-3702(87)90070-1","volume":"31","author":"D Lowe","year":"1987","unstructured":"Lowe, D. (1987). Three-dimensional object recognition from single two-dimensional images. Artificial Intelligence, 31(3), 355\u2013395.","journal-title":"Artificial Intelligence"},{"key":"780_CR31","doi-asserted-by":"crossref","unstructured":"Maji, S., & Malik, J. (2009). Object detection using a max-margin hough transform. In CVPR.","DOI":"10.1109\/CVPR.2009.5206693"},{"issue":"1","key":"780_CR32","doi-asserted-by":"crossref","first-page":"73","DOI":"10.1007\/BF00128527","volume":"1","author":"J Malik","year":"1987","unstructured":"Malik, J. (1987). Interpreting line drawings of curved objects. International Journal of Computer Vision, 1(1), 73\u2013103.","journal-title":"International Journal of Computer Vision"},{"key":"780_CR33","doi-asserted-by":"crossref","unstructured":"Meger, D., Wojek, C., Schiele, B., & Little, J. J. (2011). Explicit occlusion reasoning for 3d object detection. In BMVC.","DOI":"10.5244\/C.25.113"},{"key":"780_CR34","doi-asserted-by":"crossref","unstructured":"Oramas, J., De Raedt, L., & Tuytelaars, T. (2013). Allocentric pose estimation. In ICCV.","DOI":"10.1109\/ICCV.2013.43"},{"key":"780_CR35","doi-asserted-by":"crossref","first-page":"293","DOI":"10.1016\/0004-3702(86)90052-4","volume":"28","author":"A Pentland","year":"1986","unstructured":"Pentland, A. (1986). Perceptual organization and representation of natural form. Artificial Intelligence, 28, 293\u2013331.","journal-title":"Artificial Intelligence"},{"key":"780_CR36","doi-asserted-by":"crossref","unstructured":"Pepik, B., Stark, M., Gehler, P., & Schiele, B. (2013). Occlusion patterns for object class detection. In CVPR.","DOI":"10.1109\/CVPR.2013.422"},{"key":"780_CR37","unstructured":"Roberts, L. G. (1963) Machine perception of three-dimensional solids, Ph.D. thesis, MIT."},{"key":"780_CR38","doi-asserted-by":"crossref","unstructured":"Sch\u00f6nborn, S., Forster, A., Egger, B., & Vetter, T. (2013). A Monte Carlo strategy to integrate detection and model-based face analysis. In GCPR.","DOI":"10.1007\/978-3-642-40602-7_11"},{"key":"780_CR39","doi-asserted-by":"crossref","unstructured":"Silberman, N., Hoiem, D., Kohli, P., & Fergus, R. (2012). Indoor segmentation and support inference from RGBD images. In ECCV.","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"780_CR40","doi-asserted-by":"crossref","unstructured":"Stark, M., Goesele, M., & Schiele, B. (2010). Back to the future: Learning shape models from 3D CAD data. In BMVC.","DOI":"10.5244\/C.24.106"},{"key":"780_CR41","unstructured":"Sullivan, G. D., Worrall, A. D., & Ferryman, J. M. (1995). Visual object recognition using deformable models of vehicles. In IEEE workshop on context-based vision."},{"key":"780_CR42","unstructured":"Tang, S., Andriluka, M., & Schiele, B. (2012). Detection and tracking of occluded oeople. In BMVC."},{"key":"780_CR43","unstructured":"Vedaldi, A., & Zisserman, A. (2009). Structured output regression for detection with partial truncation. In NIPS."},{"key":"780_CR44","doi-asserted-by":"crossref","unstructured":"Villamizar, M., Grabner, H., Andrade-Cetto, J., Sanfeliu, A., Gool, L. V., & Moreno-Noguer, F. (2011). Efficient 3D object detection using multiple pose-specific classifiers. In BMVC.","DOI":"10.5244\/C.25.20"},{"key":"780_CR45","doi-asserted-by":"crossref","unstructured":"Wang, H., Gould, S., & Koller, D. (2010). Discriminative learning with latent variables for cluttered indoor scene understanding. In ECCV.","DOI":"10.1007\/978-3-642-15552-9_32"},{"key":"780_CR46","doi-asserted-by":"crossref","unstructured":"Wang, X., Han, T., & Yan, S. (2009). An HOG-LBP human detector with partial occlusion handling. In ICCV.","DOI":"10.1109\/ICCV.2009.5459207"},{"key":"780_CR47","doi-asserted-by":"crossref","unstructured":"Wojek, C., Walk, S., Roth, S., Schindler, K., & Schiele, B. (2013). Monocular visual scene understanding: Understanding multi-object traffic scenes. In PAMI.","DOI":"10.1109\/TPAMI.2012.174"},{"key":"780_CR48","doi-asserted-by":"crossref","unstructured":"Xiang, Y., & Savarese, S. (2013). Object detection by 3D aspectlets and occlusion reasoning. In 3dRR.","DOI":"10.1109\/ICCVW.2013.75"},{"key":"780_CR49","doi-asserted-by":"crossref","unstructured":"Xiang, Y., & Savarese, S. (2012). Estimating the aspect layout of object categories. In CVPR.","DOI":"10.1109\/CVPR.2012.6248081"},{"key":"780_CR50","doi-asserted-by":"crossref","unstructured":"Zhao, Y., & Zhu, S. -C. (2013). Scene parsing by integrating function, geometry and appearance models. In CVPR.","DOI":"10.1109\/CVPR.2013.401"},{"key":"780_CR51","unstructured":"Zia, M. Z., Klank, U., & Beetz, M. (2009). Acquisition of a dense 3D model database for robotic vision. In ICAR."},{"key":"780_CR52","doi-asserted-by":"crossref","unstructured":"Zia, M. Z., Stark, M., & Schindler, K. (2013). Explicit occlusion modeling for 3D object class representations. In CVPR.","DOI":"10.1109\/CVPR.2013.427"},{"key":"780_CR53","doi-asserted-by":"crossref","unstructured":"Zia, M. Z., Stark, M., & Schindler, K. (2014). Are cars just 3D boxes? Jointly estimating the 3D shape of multiple objects. In CVPR.","DOI":"10.1109\/CVPR.2014.470"},{"issue":"11","key":"780_CR54","doi-asserted-by":"crossref","first-page":"2608","DOI":"10.1109\/TPAMI.2013.87","volume":"35","author":"MZ Zia","year":"2013","unstructured":"Zia, M. Z., Stark, M., Schiele, B., & Schindler, K. (2013). Detailed 3d representations for object recognition and modeling. Pattern Analysis and Machine Intelligence, 35(11), 2608\u20132623.","journal-title":"Pattern Analysis and Machine Intelligence"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-014-0780-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11263-014-0780-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-014-0780-y","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,8,16]],"date-time":"2019-08-16T23:12:41Z","timestamp":1565997161000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11263-014-0780-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,11,4]]},"references-count":54,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2015,4]]}},"alternative-id":["780"],"URL":"https:\/\/doi.org\/10.1007\/s11263-014-0780-y","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,11,4]]}}}