{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,16]],"date-time":"2025-07-16T12:03:38Z","timestamp":1752667418117,"version":"3.37.3"},"reference-count":48,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2022,1,3]],"date-time":"2022-01-03T00:00:00Z","timestamp":1641168000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,3]],"date-time":"2022-01-03T00:00:00Z","timestamp":1641168000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2022,2]]},"DOI":"10.1007\/s11042-021-11801-3","type":"journal-article","created":{"date-parts":[[2022,1,3]],"date-time":"2022-01-03T20:03:00Z","timestamp":1641240180000},"page":"5973-5988","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Monocular 3D object detection via estimation of paired keypoints for autonomous driving"],"prefix":"10.1007","volume":"81","author":[{"given":"Chaofeng","family":"Ji","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9149-4576","authenticated-orcid":false,"given":"Guizhong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dan","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,1,3]]},"reference":[{"key":"11801_CR1","doi-asserted-by":"crossref","unstructured":"Brazil G, Liu X (2019) M3D-RPN: monocular 3d region proposal network for object detection. In: Proceedings of the IEEE international conference on computer vision, pp 9286\u20139295","DOI":"10.1109\/ICCV.2019.00938"},{"key":"11801_CR2","doi-asserted-by":"crossref","unstructured":"Cai Y, Li B, Jiao Z, Li H, Zeng X, Wang X (2020) Monocular 3d object detection with decoupled structured polygon estimation and height-guided depth estimation. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp 10478\u201310485","DOI":"10.1609\/aaai.v34i07.6618"},{"key":"11801_CR3","doi-asserted-by":"crossref","unstructured":"Cai Z, Fan Q, Feris RS, Vasconcelos N, Leibe B, Matas J, Sebe N, Welling M (2016) A unified multi-scale deep convolutional neural network for fast object detection. In: Proceedings of the European Conference on Computer Vision, pp 354\u2013370","DOI":"10.1007\/978-3-319-46493-0_22"},{"key":"11801_CR4","doi-asserted-by":"crossref","unstructured":"Chabot F, Chaouch M, Rabarisoa J, Teuli\u00e8re C, Chateau T (2017) Deep MANTA: A coarse-to-fine many-task network for joint 2d and 3d vehicle analysis from monocular image. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 1827\u20131836","DOI":"10.1109\/CVPR.2017.198"},{"key":"11801_CR5","doi-asserted-by":"crossref","unstructured":"Chang J, Chen Y (2018) Pyramid stereo matching network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 5410\u20135418","DOI":"10.1109\/CVPR.2018.00567"},{"key":"11801_CR6","doi-asserted-by":"crossref","unstructured":"Chen X, Kundu K, Zhang Z, Ma H, Fidler S, Urtasun R (2016) Monocular 3d object detection for autonomous driving. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2147\u20132156","DOI":"10.1109\/CVPR.2016.236"},{"issue":"5","key":"11801_CR7","doi-asserted-by":"publisher","first-page":"1259","DOI":"10.1109\/TPAMI.2017.2706685","volume":"40","author":"X Chen","year":"2018","unstructured":"Chen X, Kundu K, Zhu Y, Ma H, Fidler S, Urtasun R (2018) 3d object proposals using stereo imagery for accurate object class detection. IEEE Trans Pattern Anal Mach Intell 40(5):1259\u20131272","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"11801_CR8","doi-asserted-by":"crossref","unstructured":"Chen Y, Tai L, Sun K, Li M (2020) Monopair: Monocular 3d object detection using pairwise spatial relationships. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 12090\u201312099","DOI":"10.1109\/CVPR42600.2020.01211"},{"key":"11801_CR9","doi-asserted-by":"crossref","unstructured":"Ding M, Huo Y, Yi H, Wang Z, Shi J, Lu Z, Luo P (2020) Learning depth-guided convolutions for monocular 3d object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 11669\u201311678","DOI":"10.1109\/CVPR42600.2020.01169"},{"key":"11801_CR10","doi-asserted-by":"crossref","unstructured":"Duan K, Bai S, Xie L, Qi H, Huang Q, Tian Q (2019) Centernet: Keypoint triplets for object detection. In: Proceedings of the IEEE international conference on computer vision, pp 6568\u20136577","DOI":"10.1109\/ICCV.2019.00667"},{"key":"11801_CR11","doi-asserted-by":"crossref","unstructured":"Fu H, Gong M, Wang C, Batmanghelich K, Tao D (2018) Deep ordinal regression network for monocular depth estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2002\u20132011","DOI":"10.1109\/CVPR.2018.00214"},{"key":"11801_CR12","doi-asserted-by":"crossref","unstructured":"Geiger A, Lenz P, Urtasun R (2012) Are we ready for autonomous driving? the KITTI vision benchmark suite. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 3354\u20133361","DOI":"10.1109\/CVPR.2012.6248074"},{"key":"11801_CR13","doi-asserted-by":"crossref","unstructured":"Godard C, Aodha OM, Brostow GJ (2017) Unsupervised monocular depth estimation with left-right consistency. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 6602\u20136611","DOI":"10.1109\/CVPR.2017.699"},{"key":"11801_CR14","doi-asserted-by":"crossref","unstructured":"Gupta I, Rangesh A, Trivedi MM, Leal-Taix\u00e9 L, Roth S (2018) 3d bounding boxes for road vehicles: A one-stage, localization prioritized approach using single monocular images. In: Proceedings of the European Conference on Computer Vision Workshops, pp 626\u2013641","DOI":"10.1007\/978-3-030-11021-5_39"},{"key":"11801_CR15","doi-asserted-by":"crossref","unstructured":"Ku J, Pon AD, Waslander SL (2019) Monocular 3d object detection leveraging accurate proposals and shape reconstruction. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 11867\u201311876","DOI":"10.1109\/CVPR.2019.01214"},{"key":"11801_CR16","doi-asserted-by":"crossref","unstructured":"Law H, Deng J, Ferrari V, Hebert M, Sminchisescu C, Weiss Y (2018) Cornernet: Detecting objects as paired keypoints. In: Proceedings of the European Conference on Computer Vision, pp 765\u2013781","DOI":"10.1007\/978-3-030-01264-9_45"},{"key":"11801_CR17","doi-asserted-by":"crossref","unstructured":"Li B, Ouyang W, Sheng L, Zeng X, Wang X (2019a) GS3D: an efficient 3d object detection framework for autonomous driving. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 1019\u20131028","DOI":"10.1109\/CVPR.2019.00111"},{"key":"11801_CR18","doi-asserted-by":"crossref","unstructured":"Li P, Chen X, Shen S (2019b) Stereo R-CNN based 3d object detection for autonomous driving. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 7644\u20137652","DOI":"10.1109\/CVPR.2019.00783"},{"key":"11801_CR19","unstructured":"Li P, Liu S, Shen S (2019c) Multi-sensor 3d object box refinement for autonomous driving. arXiv:1909.04942"},{"key":"11801_CR20","doi-asserted-by":"crossref","unstructured":"Li P, Zhao H, Liu P, Cao F, Vedaldi A, Bischof H, Brox T, Frahm J (2020) RTM3D: real-time monocular 3d detection from object keypoints for autonomous driving. In: Proceedings of the European Conference on Computer Vision, pp 644\u2013660","DOI":"10.1007\/978-3-030-58580-8_38"},{"key":"11801_CR21","doi-asserted-by":"crossref","unstructured":"Liu L, Lu J, Xu C, Tian Q, Zhou J (2019) Deep fitting degree scoring network for monocular 3d object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 1057\u20131066","DOI":"10.1109\/CVPR.2019.00115"},{"key":"11801_CR22","doi-asserted-by":"crossref","unstructured":"Liu Z, Wu Z, T\u00f3th R (2020) SMOKE: single-stage monocular 3d object detection via keypoint estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops, pp 4289\u20134298","DOI":"10.1109\/CVPRW50498.2020.00506"},{"key":"11801_CR23","doi-asserted-by":"crossref","unstructured":"Ma X, Wang Z, Li H, Zhang P, Ouyang W, Fan X (2019) Accurate monocular 3d object detection via color-embedded 3d reconstruction for autonomous driving. In: Proceedings of the IEEE international conference on computer vision, pp 6850\u20136859","DOI":"10.1109\/ICCV.2019.00695"},{"key":"11801_CR24","doi-asserted-by":"crossref","unstructured":"Manhardt F, Kehl W, Gaidon A (2019) ROI-10D: monocular lifting of 2d detection to 6d pose and metric shape. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2069\u20132078","DOI":"10.1109\/CVPR.2019.00217"},{"key":"11801_CR25","doi-asserted-by":"crossref","unstructured":"Mousavian A, Anguelov D, Flynn J, Kosecka J (2017) 3d bounding box estimation using deep learning and geometry. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 5632\u20135640","DOI":"10.1109\/CVPR.2017.597"},{"key":"11801_CR26","first-page":"61","volume":"2019","author":"A Naiden","year":"2019","unstructured":"Naiden A, Paunescu V, Kim G, Jeon B, Leordeanu M (2019) Shift R-CNN: deep monocular 3d object detection with closed-form geometric constraints. IEEE International Conference on Image Processing, ICIP 2019:61\u201365","journal-title":"IEEE International Conference on Image Processing, ICIP"},{"key":"11801_CR27","first-page":"8383","volume":"2020","author":"AD Pon","year":"2020","unstructured":"Pon AD, Ku J, Li C, Waslander SL (2020) Object-centric stereo matching for 3d object detection. IEEE International Conference on Robotics and Automation, ICRA 2020:8383\u20138389","journal-title":"IEEE International Conference on Robotics and Automation, ICRA"},{"key":"11801_CR28","unstructured":"Qi CR, Su H, Mo K, Guibas LJ (2017) Pointnet: Deep learning on point sets for 3d classification and segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 77\u201385"},{"key":"11801_CR29","doi-asserted-by":"crossref","unstructured":"Qi CR, Liu W, Wu C, Su H, Guibas LJ (2018) Frustum pointnets for 3d object detection from RGB-D data. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 918\u2013927","DOI":"10.1109\/CVPR.2018.00102"},{"key":"11801_CR30","doi-asserted-by":"crossref","unstructured":"Qi CR, Litany O, He K, Guibas LJ (2019) Deep hough voting for 3d object detection in point clouds. In: Proceedings of the IEEE international conference on computer vision, pp 9276\u20139285","DOI":"10.1109\/ICCV.2019.00937"},{"key":"11801_CR31","doi-asserted-by":"crossref","unstructured":"Qin Z, Wang J, Lu Y (2019) Triangulation learning network: From monocular to stereo 3d object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 7615\u20137623","DOI":"10.1109\/CVPR.2019.00780"},{"key":"11801_CR32","doi-asserted-by":"crossref","unstructured":"Shi S, Wang X, Li H (2019) Pointrcnn: 3d object proposal generation and detection from point cloud. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 770\u2013779","DOI":"10.1109\/CVPR.2019.00086"},{"key":"11801_CR33","doi-asserted-by":"crossref","unstructured":"Shi S, Guo C, Jiang L, Wang Z, Shi J, Wang X, Li H (2020a) PV-RCNN: point-voxel feature set abstraction for 3d object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 10526\u201310535","DOI":"10.1109\/CVPR42600.2020.01054"},{"key":"11801_CR34","doi-asserted-by":"crossref","unstructured":"Shi X, Chen Z, Kim T (2020b) Distance-normalized unified representation for monocular 3d object detection. In: Proceedings of the European Conference on Computer Vision, pp 91\u2013107","DOI":"10.1007\/978-3-030-58526-6_6"},{"key":"11801_CR35","doi-asserted-by":"crossref","unstructured":"Simonelli A, Bul\u00f2 SR, Porzi L, Lopez-Antequera M, Kontschieder P (2019) Disentangling monocular 3d object detection. In: Proceedings of the IEEE international conference on computer vision, pp 1991\u20131999","DOI":"10.1109\/ICCV.2019.00208"},{"key":"11801_CR36","unstructured":"Simonyan K, Zisserman A, Bengio Y, LeCun Y (2015) Very deep convolutional networks for large-scale image recognition. In: 3rd International Conference on Learning Representations, ICLR 2015"},{"key":"11801_CR37","doi-asserted-by":"crossref","unstructured":"Sun J, Chen L, Xie Y, Zhang S, Jiang Q, Zhou X, Bao H (2020) Disp R-CNN: stereo 3d object detection via shape prior guided instance disparity estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 10545\u201310554","DOI":"10.1109\/CVPR42600.2020.01056"},{"key":"11801_CR38","doi-asserted-by":"crossref","unstructured":"Vora S, Lang AH, Helou B, Beijbom O (2020) Pointpainting: Sequential fusion for 3d object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 4603\u20134611","DOI":"10.1109\/CVPR42600.2020.00466"},{"key":"11801_CR39","doi-asserted-by":"crossref","unstructured":"Wang Y, Chao W, Garg D, Hariharan B, Campbell ME, Weinberger KQ (2019) Pseudo-lidar from visual depth estimation: Bridging the gap in 3d object detection for autonomous driving. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 8445\u20138453","DOI":"10.1109\/CVPR.2019.00864"},{"key":"11801_CR40","first-page":"924","volume":"2017","author":"Y Xiang","year":"2017","unstructured":"Xiang Y, Choi W, Lin Y, Savarese S (2017) Subcategory-aware convolutional neural networks for object proposals and detection. IEEE Winter Conference on Applications of Computer Vision, WACV 2017:924\u2013933","journal-title":"IEEE Winter Conference on Applications of Computer Vision, WACV"},{"key":"11801_CR41","doi-asserted-by":"crossref","unstructured":"Xie L, Xiang C, Yu Z, Xu G, Yang Z, Cai D, He X (2020) PI-RCNN: an efficient multi-sensor 3d object detector with point-based attentive cont-conv fusion module. In: Proceedings of the AAAI conference on artificial intelligence, pp 12460\u201312467","DOI":"10.1609\/aaai.v34i07.6933"},{"key":"11801_CR42","doi-asserted-by":"crossref","unstructured":"Xu B, Chen Z (2018) Multi-level fusion based 3d object detection from monocular images. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2345\u20132353","DOI":"10.1109\/CVPR.2018.00249"},{"key":"11801_CR43","doi-asserted-by":"crossref","unstructured":"Yang B, Luo W, Urtasun R (2018) PIXOR: real-time 3d object detection from point clouds. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 7652\u20137660","DOI":"10.1109\/CVPR.2018.00798"},{"key":"11801_CR44","unstructured":"Yang L, Zhang X, Wang L, Zhu M, Li J (2021) Lite-fpn for keypoint-based monocular 3d object detection. arXiv:2105.00268"},{"key":"11801_CR45","unstructured":"You Y, Wang Y, Chao W, Garg D, Pleiss G, Hariharan B, Campbell ME, Weinberger KQ (2020) Pseudo-lidar++: Accurate depth for 3d object detection in autonomous driving. In: 8th International Conference on Learning Representations, ICLR 2020"},{"key":"11801_CR46","doi-asserted-by":"crossref","unstructured":"Zhou X, Wang D, Kr\u00e4henb\u00fchl P (2019a) Objects as points. arXiv:1904.07850","DOI":"10.1007\/978-3-030-58548-8_28"},{"key":"11801_CR47","doi-asserted-by":"crossref","unstructured":"Zhou X, Zhuo J, Kr\u00e4henb\u00fchl P (2019b) Bottom-up object detection by grouping extreme and center points. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 850\u2013859","DOI":"10.1109\/CVPR.2019.00094"},{"key":"11801_CR48","doi-asserted-by":"crossref","unstructured":"Zhou Y, Tuzel O (2018) Voxelnet: End-to-end learning for point cloud based 3d object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 4490\u20134499","DOI":"10.1109\/CVPR.2018.00472"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-021-11801-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-021-11801-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-021-11801-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,15]],"date-time":"2024-09-15T16:16:38Z","timestamp":1726416998000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-021-11801-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,1,3]]},"references-count":48,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2022,2]]}},"alternative-id":["11801"],"URL":"https:\/\/doi.org\/10.1007\/s11042-021-11801-3","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"type":"print","value":"1380-7501"},{"type":"electronic","value":"1573-7721"}],"subject":[],"published":{"date-parts":[[2022,1,3]]},"assertion":[{"value":"27 July 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 October 2021","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 December 2021","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 January 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"We have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}]}}