{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T06:06:36Z","timestamp":1757311596300,"version":"3.40.3"},"publisher-location":"Cham","reference-count":41,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319249469"},{"type":"electronic","value":"9783319249476"}],"license":[{"start":{"date-parts":[[2015,1,1]],"date-time":"2015-01-01T00:00:00Z","timestamp":1420070400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc\/2.5"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2015]]},"DOI":"10.1007\/978-3-319-24947-6_43","type":"book-chapter","created":{"date-parts":[[2015,10,6]],"date-time":"2015-10-06T18:10:30Z","timestamp":1444155030000},"page":"517-528","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":34,"title":["What Is Holding Back Convnets for Detection?"],"prefix":"10.1007","author":[{"given":"Bojan","family":"Pepik","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rodrigo","family":"Benenson","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tobias","family":"Ritschel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bernt","family":"Schiele","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,11,3]]},"reference":[{"key":"43_CR1","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"329","DOI":"10.1007\/978-3-319-10584-0_22","volume-title":"Computer Vision \u2013 ECCV 2014","author":"P Agrawal","year":"2014","unstructured":"Agrawal, P., Girshick, R., Malik, J.: Analyzing the performance of multilayer neural networks for object recognition. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014, Part VII. LNCS, vol. 8695, pp. 329\u2013344. Springer, Heidelberg (2014)"},{"key":"43_CR2","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1007\/978-3-642-24412-4_3","volume-title":"Algorithmic Learning Theory","author":"Y Bengio","year":"2011","unstructured":"Bengio, Y., Delalleau, O.: On the expressive power of deep architectures. In: Kivinen, J., Szepesv\u00e1ri, C., Ukkonen, E., Zeugmann, T. (eds.) ALT 2011. LNCS, vol. 6925, pp. 18\u201336. Springer, Heidelberg (2011)"},{"key":"43_CR3","doi-asserted-by":"crossref","unstructured":"Chatfield, K., Simonyan, K., Vedaldi, A., Zisserman, A.: Return of the devil in the details: delving deep into convolutional nets. In: BMVC (2014)","DOI":"10.5244\/C.28.6"},{"key":"43_CR4","unstructured":"Chen, X., Yuille, A.: Articulated pose estimation by a graphical model with image dependent pairwise relations. In: NIPS (2014)"},{"key":"43_CR5","unstructured":"Dauphin, Y.N., Pascanu, R., Gulcehre, C., Cho, K., Ganguli, S., Bengio, Y.: Identifying and attacking the saddle point problem in high-dimensional non-convex optimization. In: NIPS, pp. 2933\u20132941 (2014)"},{"key":"43_CR6","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: ImageNet: a large-scale hierarchical image database. In: CVPR (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"43_CR7","doi-asserted-by":"crossref","unstructured":"Enzweiler, M., Gavrila, D.M.: A mixed generative-discriminative framework for pedestrian classification. In: CVPR, pp. 1\u20138. IEEE (2008)","DOI":"10.1109\/CVPR.2008.4587592"},{"key":"43_CR8","volume-title":"The 2007 Pascal Visual Object Classes Challenge","author":"M Everingham","year":"2007","unstructured":"Everingham, M., Zisserman, A., Williams, C.K.I., Van Gool, L.: The 2007 Pascal Visual Object Classes Challenge. Springer-Verlag, Berlin (2007)"},{"key":"43_CR9","doi-asserted-by":"crossref","unstructured":"Fischer, P., Dosovitskiy, A., Ilg, E., H\u00e4usser, P., Hazirbas, C., Golkov, V., van der Smagt, P., Cremers, D., Brox, T.: Flownet: learning optical flow with convolutional networks. Arxiv. No. 1405.5769 (2015). http:\/\/lmb.informatik.uni-freiburg.de\/\/Publications\/2015\/FDIB15","DOI":"10.1109\/ICCV.2015.316"},{"key":"43_CR10","doi-asserted-by":"crossref","unstructured":"Fischer, P., Dosovitskiy, A., Brox, T.: Descriptor matching with convolutional neural networks: a comparison to sift (2014). arXiv:1405.5769","DOI":"10.1109\/CVPR.2015.7298761"},{"key":"43_CR11","doi-asserted-by":"crossref","unstructured":"Girshick, R.: Fast R-CNN (2015). arXiv:1504.08083","DOI":"10.1109\/ICCV.2015.169"},{"key":"43_CR12","doi-asserted-by":"crossref","unstructured":"Girshick, R., Donahue, J., Darrell, T., Malik, J.: Rich feature hierarchies for accurate object detection and semantic segmentation. arXiv (2014)","DOI":"10.1109\/CVPR.2014.81"},{"key":"43_CR13","unstructured":"Goodfellow, I., Le, Q., Saxe, A., Ng, A.Y.: Measuring invariances in deep networks. In: NIPS (2009)"},{"key":"43_CR14","unstructured":"Goodfellow, I.J., Shlens, J., Szegedy, C.: Explaining and harnessing adversarial examples. In: ICLR (2015)"},{"key":"43_CR15","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"340","DOI":"10.1007\/978-3-642-33712-3_25","volume-title":"Computer Vision \u2013 ECCV 2012","author":"D Hoiem","year":"2012","unstructured":"Hoiem, D., Chodpathumwan, Y., Dai, Q.: Diagnosing error in object detectors. In: Fitzgibbon, A., Lazebnik, S., Perona, P., Sato, Y., Schmid, C. (eds.) ECCV 2012, Part III. LNCS, vol. 7574, pp. 340\u2013353. Springer, Heidelberg (2012)"},{"key":"43_CR16","unstructured":"Ioffe, S., Szegedy, C.: Batch normalization: accelerating deep network training by reducing internal covariate shift (2015). arXiv:1502.03167"},{"key":"43_CR17","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: NIPS (2012)"},{"key":"43_CR18","doi-asserted-by":"crossref","unstructured":"Le, Q.V., Monga, R., Devin, M., Chen, K., Corrado, G.S., Dean, J., Ng, A.Y.: Building high-level features using large scale unsupervised learning. In: ICML (2012)","DOI":"10.1109\/ICASSP.2013.6639343"},{"key":"43_CR19","doi-asserted-by":"crossref","unstructured":"Lenc, K., Vedaldi, A.: Understanding image representations by measuring their equivariance and equivalence. In: CVPR (2015)","DOI":"10.1109\/CVPR.2015.7298701"},{"key":"43_CR20","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"194","DOI":"10.1007\/978-3-319-16814-2_13","volume-title":"Computer Vision \u2013 ACCV 2014","author":"H Li","year":"2015","unstructured":"Li, H., Li, Y., Porikli, F.: Robust online visual tracking with a single convolutional neural network. In: Cremers, D., Reid, I., Saito, H., Yang, M.-H. (eds.) ACCV 2014. LNCS, vol. 9007, pp. 194\u2013209. Springer, Heidelberg (2015)"},{"key":"43_CR21","doi-asserted-by":"crossref","unstructured":"Long, J., Shelhamer, E., Darrell, T.: Fully convolutional networks for semantic segmentation. In: CVPR, November 2015","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"43_CR22","doi-asserted-by":"crossref","unstructured":"Mahendran, A., Vedaldi, A.: Understanding deep image representations by inverting them. In: CVPR, June 2015","DOI":"10.1109\/CVPR.2015.7299155"},{"key":"43_CR23","doi-asserted-by":"crossref","unstructured":"Pepik, B., Stark, M., Gehler, P., Ritschel, T., Schiele, B.: 3D object class detection in the wild. In: 3DSI in Conjunction with CVPR (2015)","DOI":"10.1109\/CVPRW.2015.7301358"},{"key":"43_CR24","doi-asserted-by":"crossref","unstructured":"Pepik, B., Stark, M., Gehler, P., Schiele, B.: Multi-view and 3D deformable part models. TPAMI (2015)","DOI":"10.1109\/TPAMI.2015.2408347"},{"key":"43_CR25","doi-asserted-by":"crossref","unstructured":"Pishchulin, L., Jain, A., Andriluka, M., Thormaehlen, T., Schiele, B.: Articulated people detection and pose estimation: reshaping the future. In: CVPR, June 2012","DOI":"10.1109\/CVPR.2012.6248052"},{"key":"43_CR26","doi-asserted-by":"crossref","unstructured":"Razavian, A.S., Azizpour, H., Maki, A., Sullivan, J., Ek, C.H., Carlsson, S.: Persistent evidence of local image properties in generic convnets (2014). arXiv:1411.6509","DOI":"10.1007\/978-3-319-19665-7_21"},{"key":"43_CR27","doi-asserted-by":"crossref","unstructured":"Razavian, A.S., Azizpour, H., Sullivan, J., Carlsson, S.: CNN features off-the-shelf: an astounding baseline for recognition. In: CVPR Workshops, pp. 512\u2013519. IEEE (2014)","DOI":"10.1109\/CVPRW.2014.131"},{"key":"43_CR28","doi-asserted-by":"crossref","unstructured":"Schroff, F., Kalenichenko, D., Philbin, J.: Facenet: A unified embedding for face recognition and clustering (2015). arXiv:1503.03832","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"43_CR29","unstructured":"Simonyan, K., Vedaldi, A., Zisserman, A.: Deep inside convolutional networks: visualising image classification models and saliency maps. In: ICLR Workshop (2014)"},{"key":"43_CR30","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. In: ICLR (2015)"},{"key":"43_CR31","unstructured":"Springenberg, J.T., Dosovitskiy, A., Brox, T., Riedmiller, M.: Striving for simplicity: the all convolutional net. In: ICLR (2015)"},{"key":"43_CR32","doi-asserted-by":"crossref","unstructured":"Stark, M., Goesele, M., Schiele, B.: Back to the future: learning shape models from 3D CAD data. In: BMVC, vol. 2, p. 5 (2010)","DOI":"10.5244\/C.24.106"},{"key":"43_CR33","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Liu, W., Jia, Y., Sermanet, P., Reed, S., Anguelov, D., Erhan, D., Vanhoucke, V., Rabinovich, A.: Going deeper with convolutions (2014). arXiv preprint arXiv:1409.4842","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"43_CR34","unstructured":"Szegedy, C., Zaremba, W., Sutskever, I., Bruna, J., Erhan, D., Goodfellow, I., Fergus, R.: Intriguing properties of neural networks. In: ICLR (2014)"},{"key":"43_CR35","doi-asserted-by":"crossref","unstructured":"Torralba, A., Efros, A.A.: Unbiased look at dataset bias. In: CVPR, pp. 1521\u20131528. IEEE (2011)","DOI":"10.1109\/CVPR.2011.5995347"},{"key":"43_CR36","doi-asserted-by":"crossref","unstructured":"Uijlings, J., van de Sande, K., Gevers, T., Smeulders, A.: Selective search for object recognition. In: IJCV (2013)","DOI":"10.1007\/s11263-013-0620-5"},{"key":"43_CR37","doi-asserted-by":"crossref","unstructured":"Xiang, Y., Mottaghi, R., Savarese, S.: Beyond pascal: a benchmark for 3D object detection in the wild. In: WACV (2014)","DOI":"10.1109\/WACV.2014.6836101"},{"key":"43_CR38","doi-asserted-by":"crossref","unstructured":"Xie, S., Tu, Z.: Holistically-nested edge detection (2015). arXiv:1504.06375","DOI":"10.1109\/ICCV.2015.164"},{"issue":"5","key":"43_CR39","doi-asserted-by":"publisher","first-page":"2121","DOI":"10.1109\/TITS.2014.2310138","volume":"15","author":"J Xu","year":"2014","unstructured":"Xu, J., Vazquez, D., Lopez, A.M., Marin, J., Ponsa, D.: Learning a part-based pedestrian detector in a virtual world. IEEE Trans. Intell. Transp. Syst. 15(5), 2121\u20132131 (2014)","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"43_CR40","doi-asserted-by":"crossref","unstructured":"Zbontar, J., LeCun, Y.: Computing the stereo matching cost with a convolutional neural network. In: CVPR, June 2015","DOI":"10.1109\/CVPR.2015.7298767"},{"key":"43_CR41","doi-asserted-by":"crossref","unstructured":"Zhu, X., Vondrick, C., Ramanan, D., Fowlkes, C.: Do we need more training data or better models for object detection? In: BMVC (2012)","DOI":"10.5244\/C.26.80"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-24947-6_43","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,12,28]],"date-time":"2023-12-28T09:04:18Z","timestamp":1703754258000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-319-24947-6_43"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015]]},"ISBN":["9783319249469","9783319249476"],"references-count":41,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-24947-6_43","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2015]]},"assertion":[{"value":"3 November 2015","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}