{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T22:04:24Z","timestamp":1783807464218,"version":"3.55.0"},"publisher-location":"Cham","reference-count":80,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729829","type":"print"},{"value":"9783031729836","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T00:00:00Z","timestamp":1730160000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T00:00:00Z","timestamp":1730160000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72983-6_17","type":"book-chapter","created":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T09:34:20Z","timestamp":1730108060000},"page":"290-309","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["General Geometry-Aware Weakly Supervised 3D Object Detection"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6692-1185","authenticated-orcid":false,"given":"Guowen","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6989-2711","authenticated-orcid":false,"given":"Junsong","family":"Fan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6600-5064","authenticated-orcid":false,"given":"Liyi","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2648-3875","authenticated-orcid":false,"given":"Zhaoxiang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0791-189X","authenticated-orcid":false,"given":"Zhen","family":"Lei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2078-4215","authenticated-orcid":false,"given":"Lei","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,29]]},"reference":[{"issue":"6","key":"17_CR1","doi-asserted-by":"publisher","first-page":"641","DOI":"10.1109\/34.295913","volume":"16","author":"R Adams","year":"1994","unstructured":"Adams, R., Bischof, L.: Seeded region growing. IEEE Trans. Pattern Anal. Mach. Intell. 16(6), 641\u2013647 (1994)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"17_CR2","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., et al.: Language models are few-shot learners. Adv. Neural. Inf. Process. Syst. 33, 1877\u20131901 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"17_CR3","doi-asserted-by":"crossref","unstructured":"Caesar, H., et al.: Nuscenes: a multimodal dataset for autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11621\u201311631 (2020)","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"17_CR4","unstructured":"Chang, A.X., et\u00a0al.: ShapeNet: an information-rich 3D model repository. arXiv preprint arXiv:1512.03012 (2015)"},{"key":"17_CR5","doi-asserted-by":"crossref","unstructured":"Chen, H., Huang, Y., Tian, W., Gao, Z., Xiong, L.: Monorun: monocular 3D object detection by reconstruction and uncertainty propagation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10379\u201310388 (2021)","DOI":"10.1109\/CVPR46437.2021.01024"},{"key":"17_CR6","doi-asserted-by":"crossref","unstructured":"Chen, L., Lei, C., Li, R., Li, S., Zhang, Z., Zhang, L.: FPR: false positive rectification for weakly supervised semantic segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1108\u20131118 (2023)","DOI":"10.1109\/ICCV51070.2023.00108"},{"key":"17_CR7","doi-asserted-by":"crossref","unstructured":"Chen, X., Ma, H., Wan, J., Li, B., Xia, T.: Multi-view 3D object detection network for autonomous driving. In: Proceedings of the IEEE conference on Computer Vision and Pattern Recognition, pp. 1907\u20131915 (2017)","DOI":"10.1109\/CVPR.2017.691"},{"key":"17_CR8","doi-asserted-by":"crossref","unstructured":"Cheng, B., Sheng, L., Shi, S., Yang, M., Xu, D.: Back-tracing representative points for voting-based 3d object detection in point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8963\u20138972 (2021)","DOI":"10.1109\/CVPR46437.2021.00885"},{"key":"17_CR9","unstructured":"Contributors, M.: MMDetection3D: OpenMMLab next-generation platform for general 3D object detection (2020). https:\/\/github.com\/open-mmlab\/mmdetection3d"},{"key":"17_CR10","doi-asserted-by":"crossref","unstructured":"Ding, M., et al.: Learning depth-guided convolutions for monocular 3D object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, pp. 1000\u20131001 (2020)","DOI":"10.1109\/CVPRW50498.2020.00508"},{"key":"17_CR11","doi-asserted-by":"crossref","unstructured":"Fan, L., et al. Embracing single stride 3D object detector with sparse transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8458\u20138468 (2022)","DOI":"10.1109\/CVPR52688.2022.00827"},{"key":"17_CR12","doi-asserted-by":"crossref","unstructured":"Fan, L., Xiong, X., Wang, F., Wang, N., Zhang, Z.: Rangedet: in defense of range view for lidar-based 3D object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2918\u20132927 (2021)","DOI":"10.1109\/ICCV48922.2021.00291"},{"issue":"6","key":"17_CR13","doi-asserted-by":"publisher","first-page":"381","DOI":"10.1145\/358669.358692","volume":"24","author":"MA Fischler","year":"1981","unstructured":"Fischler, M.A., Bolles, R.C.: Random sample consensus: a paradigm for model fitting with applications to image analysis and automated cartography. Commun. ACM 24(6), 381\u2013395 (1981)","journal-title":"Commun. ACM"},{"key":"17_CR14","doi-asserted-by":"crossref","unstructured":"Geiger, A., Lenz, P., Urtasun, R.: Are we ready for autonomous driving? the Kitti vision benchmark suite. In: 2012 IEEE Conference on Computer Vision and Pattern Recognition, pp. 3354\u20133361. IEEE (2012)","DOI":"10.1109\/CVPR.2012.6248074"},{"key":"17_CR15","doi-asserted-by":"crossref","unstructured":"He, C., Li, R., Zhang, G., Zhang, L.: ScatterFormer: efficient voxel transformer with scattered linear attention. arXiv preprint arXiv:2401.00912 (2024)","DOI":"10.1007\/978-3-031-73397-0_5"},{"key":"17_CR16","doi-asserted-by":"crossref","unstructured":"He, C., Li, R., Zhang, Y., Li, S., Zhang, L.: MSF: motion-guided sequential fusion for efficient 3D object detection from point cloud sequences. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5196\u20135205 (2023)","DOI":"10.1109\/CVPR52729.2023.00503"},{"key":"17_CR17","doi-asserted-by":"crossref","unstructured":"He, C., Zeng, H., Huang, J., Hua, X.S., Zhang, L.: Structure aware single-stage 3D object detection from point cloud. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11873\u201311882 (2020)","DOI":"10.1109\/CVPR42600.2020.01189"},{"key":"17_CR18","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"17_CR19","doi-asserted-by":"crossref","unstructured":"Hou, J., Dai, A., Nie\u00dfner, M.: 3D-sis: 3D semantic instance segmentation of RGB-D scans. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 4421\u20134430 (2019)","DOI":"10.1109\/CVPR.2019.00455"},{"key":"17_CR20","doi-asserted-by":"crossref","unstructured":"Hu, Y., et\u00a0al.: Planning-oriented autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17853\u201317862 (2023)","DOI":"10.1109\/CVPR52729.2023.01712"},{"key":"17_CR21","unstructured":"Huang, J., et al.: An embodied generalist agent in 3D world. arXiv preprint arXiv:2311.12871 (2023)"},{"key":"17_CR22","doi-asserted-by":"crossref","unstructured":"Huang, K.C., Wu, T.H., Su, H.T., Hsu, W.H.: MonoDTR: monocular 3D object detection with depth-aware transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4012\u20134021 (2022)","DOI":"10.1109\/CVPR52688.2022.00398"},{"issue":"10","key":"17_CR23","doi-asserted-by":"publisher","first-page":"2702","DOI":"10.1109\/TPAMI.2019.2926463","volume":"42","author":"X Huang","year":"2019","unstructured":"Huang, X., Wang, P., Cheng, X., Zhou, D., Geng, Q., Yang, R.: The apolloscape open dataset for autonomous driving and its application. IEEE Trans. Pattern Anal. Mach. Intell. 42(10), 2702\u20132719 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"17_CR24","unstructured":"Ke, L., et al.: Segment anything in high quality. arXiv preprint arXiv:2306.01567 (2023)"},{"key":"17_CR25","unstructured":"Kirillov, A., et\u00a0al.: Segment anything. arXiv preprint arXiv:2304.02643 (2023)"},{"key":"17_CR26","doi-asserted-by":"crossref","unstructured":"Ku, J., Pon, A.D., Waslander, S.L.: Monocular 3D object detection leveraging accurate proposals and shape reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11867\u201311876 (2019)","DOI":"10.1109\/CVPR.2019.01214"},{"key":"17_CR27","doi-asserted-by":"crossref","unstructured":"Lang, A.H., Vora, S., Caesar, H., Zhou, L., Yang, J., Beijbom, O.: PointPillars: fast encoders for object detection from point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12697\u201312705 (2019)","DOI":"10.1109\/CVPR.2019.01298"},{"key":"17_CR28","doi-asserted-by":"crossref","unstructured":"Li, B., Ouyang, W., Sheng, L., Zeng, X., Wang, X.: GS3D: an efficient 3d object detection framework for autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1019\u20131028 (2019)","DOI":"10.1109\/CVPR.2019.00111"},{"key":"17_CR29","doi-asserted-by":"publisher","first-page":"718","DOI":"10.1007\/978-3-031-20077-9_42","volume-title":"ECCV 2022","author":"Y Li","year":"2022","unstructured":"Li, Y., Chen, Y., He, J., Zhang, Z.: Densely constrained depth estimator for monocular 3d object detection. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022, vol. 13669, pp. 718\u2013734. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20077-9_42"},{"key":"17_CR30","doi-asserted-by":"crossref","unstructured":"Li, Z., Wang, F., Wang, N.: Lidar R-CNN: an efficient and universal 3D object detector. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7546\u20137555 (2021)","DOI":"10.1109\/CVPR46437.2021.00746"},{"key":"17_CR31","doi-asserted-by":"crossref","unstructured":"Liang, Z., Zhang, Z., Zhang, M., Zhao, X., Pu, S.: Rangeioudet: range image based real-time 3D object detector optimized by intersection over union. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7140\u20137149 (2021)","DOI":"10.1109\/CVPR46437.2021.00706"},{"key":"17_CR32","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"657","DOI":"10.1007\/978-3-031-19839-7_38","volume-title":"ECCV 2022","author":"C Liu","year":"2022","unstructured":"Liu, C., Qian, X., Huang, B., Qi, X., Lam, E., Tan, S.C., Wong, N.: Multimodal transformer for automatic 3d annotation and object detection. In: ECCV 2022. LNCS, vol. 13698, pp. 657\u2013673. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19839-7_38"},{"key":"17_CR33","doi-asserted-by":"crossref","unstructured":"Liu, L., Lu, J., Xu, C., Tian, Q., Zhou, J.: Deep fitting degree scoring network for monocular 3D object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1057\u20131066 (2019)","DOI":"10.1109\/CVPR.2019.00115"},{"key":"17_CR34","doi-asserted-by":"crossref","unstructured":"Liu, Z., Zhang, Z., Cao, Y., Hu, H., Tong, X.: Group-free 3D object detection via transformers. 2021 IEEE. In: CVF International Conference on Computer Vision (ICCV), pp. 2929\u20132938 (2021)","DOI":"10.1109\/ICCV48922.2021.00294"},{"key":"17_CR35","unstructured":"Liu, Z., Tang, H., Lin, Y., Han, S.: Point-voxel CNN for efficient 3D deep learning. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"17_CR36","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"311","DOI":"10.1007\/978-3-030-58601-0_19","volume-title":"Computer Vision \u2013 ECCV 2020","author":"X Ma","year":"2020","unstructured":"Ma, X., Liu, S., Xia, Z., Zhang, H., Zeng, X., Ouyang, W.: Rethinking pseudo-LiDAR representation. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020, Part VIII. LNCS, vol. 12358, pp. 311\u2013327. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58601-0_19"},{"key":"17_CR37","doi-asserted-by":"crossref","unstructured":"Manhardt, F., Kehl, W., Gaidon, A.: Roi-10D: monocular lifting of 2D detection to 6D pose and metric shape. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2069\u20132078 (2019)","DOI":"10.1109\/CVPR.2019.00217"},{"key":"17_CR38","doi-asserted-by":"crossref","unstructured":"McCraith, R., Insafutdinov, E., Neumann, L., Vedaldi, A.: Lifting 2D object locations to 3D by discounting lidar outliers across objects and views. In: 2022 International Conference on Robotics and Automation (ICRA), pp. 2411\u20132418. IEEE (2022)","DOI":"10.1109\/ICRA46639.2022.9811693"},{"key":"17_CR39","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"515","DOI":"10.1007\/978-3-030-58601-0_31","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Q Meng","year":"2020","unstructured":"Meng, Q., Wang, W., Zhou, T., Shen, J., Van Gool, L., Dai, D.: Weakly supervised 3D object detection from lidar point cloud. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12358, pp. 515\u2013531. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58601-0_31"},{"key":"17_CR40","doi-asserted-by":"crossref","unstructured":"Meyer, G.P., Laddha, A., Kee, E., Vallespi-Gonzalez, C., Wellington, C.K.: LaserNet: an efficient probabilistic 3D object detector for autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12677\u201312686 (2019)","DOI":"10.1109\/CVPR.2019.01296"},{"key":"17_CR41","doi-asserted-by":"crossref","unstructured":"Misra, I., Girdhar, R., Joulin, A.: An end-to-end transformer model for 3D object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2906\u20132917 (2021)","DOI":"10.1109\/ICCV48922.2021.00290"},{"key":"17_CR42","doi-asserted-by":"crossref","unstructured":"Mousavian, A., Anguelov, D., Flynn, J., Kosecka, J.: 3D bounding box estimation using deep learning and geometry. In: Proceedings of the IEEE conference on Computer Vision and Pattern Recognition, pp. 7074\u20137082 (2017)","DOI":"10.1109\/CVPR.2017.597"},{"key":"17_CR43","doi-asserted-by":"crossref","unstructured":"Papadopoulos, D.P., Uijlings, J.R., Keller, F., Ferrari, V.: Extreme clicking for efficient object annotation. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4930\u20134939 (2017)","DOI":"10.1109\/ICCV.2017.528"},{"key":"17_CR44","doi-asserted-by":"crossref","unstructured":"Park, J.J., Florence, P., Straub, J., Newcombe, R., Lovegrove, S.: DeepsDF: learning continuous signed distance functions for shape representation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 165\u2013174 (2019)","DOI":"10.1109\/CVPR.2019.00025"},{"key":"17_CR45","unstructured":"Peng, L., Yan, S., Wu, B., Yang, Z., He, X., Cai, D.: Weakm3D: towards weakly supervised monocular 3d object detection. arXiv preprint arXiv:2203.08332 (2022)"},{"key":"17_CR46","doi-asserted-by":"crossref","unstructured":"Qi, C.R., Litany, O., He, K., Guibas, L.J.: Deep hough voting for 3D object detection in point clouds. In: proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9277\u20139286 (2019)","DOI":"10.1109\/ICCV.2019.00937"},{"key":"17_CR47","doi-asserted-by":"crossref","unstructured":"Qi, C.R., Liu, W., Wu, C., Su, H., Guibas, L.J.: Frustum PointNets for 3D object detection from RGB-D data. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 918\u2013927 (2018)","DOI":"10.1109\/CVPR.2018.00102"},{"key":"17_CR48","unstructured":"Qi, C.R., Su, H., Mo, K., Guibas, L.J.: Pointnet: deep learning on point sets for 3D classification and segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 652\u2013660 (2017)"},{"key":"17_CR49","unstructured":"Qi, C.R., Yi, L., Su, H., Guibas, L.J.: PointNet++: deep hierarchical feature learning on point sets in a metric space. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"17_CR50","doi-asserted-by":"crossref","unstructured":"Qin, Z., Wang, J., Lu, Y.: Weakly supervised 3D object detection from point clouds. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 4144\u20134152 (2020)","DOI":"10.1145\/3394171.3413805"},{"key":"17_CR51","unstructured":"Roddick, T., Kendall, A., Cipolla, R.: Orthographic feature transform for monocular 3D object detection. arXiv preprint arXiv:1811.08188 (2018)"},{"key":"17_CR52","series-title":"LNCS","first-page":"477","volume-title":"ECCV 2022","author":"D Rukhovich","year":"2022","unstructured":"Rukhovich, D., Vorontsova, A., Konushin, A.: FCAF3D: fully convolutional anchor-free 3d object detection. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13670, pp. 477\u2013493. Springer, Cham (2022)"},{"key":"17_CR53","doi-asserted-by":"crossref","unstructured":"Shen, X., Stamos, I.: Frustum voxnet for 3d object detection from RGB-D or depth images. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1698\u20131706 (2020)","DOI":"10.1109\/WACV45572.2020.9093276"},{"key":"17_CR54","doi-asserted-by":"crossref","unstructured":"Shi, S., et al.: PV-RCNN: point-voxel feature set abstraction for 3D object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10529\u201310538 (2020)","DOI":"10.1109\/CVPR42600.2020.01054"},{"issue":"2","key":"17_CR55","doi-asserted-by":"publisher","first-page":"531","DOI":"10.1007\/s11263-022-01710-9","volume":"131","author":"S Shi","year":"2023","unstructured":"Shi, S., et al.: PV-RCNN++: point-voxel feature set abstraction with local vector representation for 3d object detection. Int. J. Comput. Vision 131(2), 531\u2013551 (2023)","journal-title":"Int. J. Comput. Vision"},{"key":"17_CR56","doi-asserted-by":"crossref","unstructured":"Shi, S., Wang, X., Li, H.: Pointrcnn: 3d object proposal generation and detection from point cloud. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 770\u2013779 (2019)","DOI":"10.1109\/CVPR.2019.00086"},{"key":"17_CR57","doi-asserted-by":"crossref","unstructured":"Simonelli, A., Bulo, S.R., Porzi, L., L\u00f3pez-Antequera, M., Kontschieder, P.: Disentangling monocular 3D object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1991\u20131999 (2019)","DOI":"10.1109\/ICCV.2019.00208"},{"key":"17_CR58","doi-asserted-by":"crossref","unstructured":"Song, S., Lichtenberg, S.P., Xiao, J.: Sun RGB-D: A RGB-D scene understanding benchmark suite. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 567\u2013576 (2015)","DOI":"10.1109\/CVPR.2015.7298655"},{"key":"17_CR59","doi-asserted-by":"crossref","unstructured":"Song, S., Xiao, J.: Deep sliding shapes for Amodal 3D object detection in RGB-D images. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 808\u2013816 (2016)","DOI":"10.1109\/CVPR.2016.94"},{"key":"17_CR60","doi-asserted-by":"crossref","unstructured":"Sun, P., et\u00a0al.: Scalability in perception for autonomous driving: waymo open dataset. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2446\u20132454 (2020)","DOI":"10.1109\/CVPR42600.2020.00252"},{"key":"17_CR61","doi-asserted-by":"crossref","unstructured":"Tang, Y.S., Lee, G.H.: Transferable semi-supervised 3D object detection from RGB-D data. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1931\u20131940 (2019)","DOI":"10.1109\/ICCV.2019.00202"},{"key":"17_CR62","doi-asserted-by":"crossref","unstructured":"Tang, Y.S., Lee, G.H.: Transferable semi-supervised 3D object detection from RGB-D data. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1931\u20131940 (2019)","DOI":"10.1109\/ICCV.2019.00202"},{"key":"17_CR63","doi-asserted-by":"crossref","unstructured":"Tao, R., Han, W., Qiu, Z., Xu, C.Z., Shen, J.: Weakly supervised monocular 3D object detection using multi-view projection and direction consistency. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17482\u201317492 (2023)","DOI":"10.1109\/CVPR52729.2023.01677"},{"key":"17_CR64","unstructured":"Team, G., et\u00a0al.: Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)"},{"key":"17_CR65","doi-asserted-by":"crossref","unstructured":"Tian, Z., Shen, C., Chen, H., He, T.: FCOS: fully convolutional one-stage object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9627\u20139636 (2019)","DOI":"10.1109\/ICCV.2019.00972"},{"key":"17_CR66","unstructured":"Touvron, H., et\u00a0al.: Llama: open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)"},{"key":"17_CR67","unstructured":"Wang, T., Xinge, Z., Pang, J., Lin, D.: Probabilistic and geometric depth: detecting objects in perspective. In: Conference on Robot Learning, pp. 1475\u20131485. PMLR (2022)"},{"key":"17_CR68","unstructured":"Wang, Y., Chen, Y., Zhang, Z.X.: 4D unsupervised object discovery. In: Advances in Neural Information Processing Systems, vol. 35, pp. 35563\u201335575 (2022)"},{"key":"17_CR69","doi-asserted-by":"crossref","unstructured":"Wang, Y., He, J., Fan, L., Li, H., Chen, Y., Zhang, Z.: Driving into the future: Multiview visual forecasting and planning with world model for autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14749\u201314759 (2024)","DOI":"10.1109\/CVPR52733.2024.01397"},{"key":"17_CR70","doi-asserted-by":"crossref","unstructured":"Wei, Y., Su, S., Lu, J., Zhou, J.: FGR: frustum-aware geometric reasoning for weakly supervised 3Dvehicle detection. In: 2021 IEEE International Conference on Robotics and Automation (ICRA), pp. 4348\u20134354. IEEE (2021)","DOI":"10.1109\/ICRA48506.2021.9561245"},{"key":"17_CR71","doi-asserted-by":"crossref","unstructured":"Xie, Q.,et al.: MLCVNET: multi-level context votenet for 3D object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10447\u201310456 (2020)","DOI":"10.1109\/CVPR42600.2020.01046"},{"issue":"10","key":"17_CR72","doi-asserted-by":"publisher","first-page":"3337","DOI":"10.3390\/s18103337","volume":"18","author":"Y Yan","year":"2018","unstructured":"Yan, Y., Mao, Y., Li, B.: Second: sparsely embedded convolutional detection. Sensors 18(10), 3337 (2018)","journal-title":"Sensors"},{"key":"17_CR73","doi-asserted-by":"crossref","unstructured":"Yin, T., Zhou, X., Krahenbuhl, P.: Center-based 3D object detection and tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11784\u201311793 (2021)","DOI":"10.1109\/CVPR46437.2021.01161"},{"key":"17_CR74","doi-asserted-by":"crossref","unstructured":"Zakharov, S., Kehl, W., Bhargava, A., Gaidon, A.: Autolabeling 3D objects with differentiable rendering of SDF shape priors. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12224\u201312233 (2020)","DOI":"10.1109\/CVPR42600.2020.01224"},{"key":"17_CR75","doi-asserted-by":"crossref","unstructured":"Zakharov, S., Kehl, W., Bhargava, A., Gaidon, A.: Autolabeling 3D objects with differentiable rendering of SDF shape priors. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12224\u201312233 (2020)","DOI":"10.1109\/CVPR42600.2020.01224"},{"key":"17_CR76","unstructured":"Zhang, G., Fan, L., He, C., Lei, Z., Zhang, Z., Zhang, L.: Voxel mamba: group-free state space models for point cloud based 3D object detection. arXiv preprint arXiv:2406.10700 (2024)"},{"key":"17_CR77","doi-asserted-by":"crossref","unstructured":"Zhang, R., et al.: Monodetr: depth-guided transformer for monocular 3D object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9155\u20139166 (2023)","DOI":"10.1109\/ICCV51070.2023.00840"},{"key":"17_CR78","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"311","DOI":"10.1007\/978-3-030-58610-2_19","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Z Zhang","year":"2020","unstructured":"Zhang, Z., Sun, B., Yang, H., Huang, Q.: H3DNet: 3D object detection using hybrid geometric primitives. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12357, pp. 311\u2013329. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58610-2_19"},{"key":"17_CR79","unstructured":"Zhou, B., Lapedriza, A., Xiao, J., Torralba, A., Oliva, A.: Learning deep features for scene recognition using places database. In: Advances in Neural Information Processing Systems, vol. 27 (2014)"},{"key":"17_CR80","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Tuzel, O.: VoxelNet: end-to-end learning for point cloud based 3D object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4490\u20134499 (2018)","DOI":"10.1109\/CVPR.2018.00472"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72983-6_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,30]],"date-time":"2024-11-30T10:33:34Z","timestamp":1732962814000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72983-6_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,29]]},"ISBN":["9783031729829","9783031729836"],"references-count":80,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72983-6_17","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,29]]},"assertion":[{"value":"29 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}