{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T03:45:01Z","timestamp":1774583101440,"version":"3.50.1"},"reference-count":66,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2025,8,25]],"date-time":"2025-08-25T00:00:00Z","timestamp":1756080000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,8,25]],"date-time":"2025-08-25T00:00:00Z","timestamp":1756080000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100018527","name":"Key Research Program of Frontier Science, Chinese Academy of Sciences","doi-asserted-by":"publisher","award":["ZDBS-LY-JSC00"],"award-info":[{"award-number":["ZDBS-LY-JSC00"]}],"id":[{"id":"10.13039\/501100018527","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,11]]},"DOI":"10.1007\/s11263-025-02544-x","type":"journal-article","created":{"date-parts":[[2025,8,25]],"date-time":"2025-08-25T15:58:15Z","timestamp":1756137495000},"page":"8022-8040","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["OA-DET3D: Embedding Object Awareness As A General Plug-in for Multi-Camera 3D Object Detection"],"prefix":"10.1007","volume":"133","author":[{"given":"Xiaomeng","family":"Chu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiajun","family":"Deng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianmin","family":"Ji","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Houqiang","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6520-255X","authenticated-orcid":false,"given":"Yanyong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,8,25]]},"reference":[{"key":"2544_CR1","doi-asserted-by":"crossref","unstructured":"Brazil, G., &Liu, X. (2019). M3d-rpn: Monocular 3d region proposal network for object detection. In Proceedings of the IEEE\/CVF International Conference on Computer Cision, (pp 9287\u20139296).","DOI":"10.1109\/ICCV.2019.00938"},{"key":"2544_CR2","doi-asserted-by":"crossref","unstructured":"Caesar, H., Bankiti, V., Lang, A. H., Vora, S., Liong, V. E., Xu, Q., Krishnan, A., Pan, Y., Baldan, G., & Beijbom, O. (2020). nuscenes: A multimodal dataset for autonomous driving. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp 11621\u201311631).","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"2544_CR3","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., & Zagoruyko, S. (2020). End-to-end object detection with transformers., In European Conference on Computer Vision, Springer, 213\u2013229.","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"2544_CR4","doi-asserted-by":"crossref","unstructured":"Chu, X., Deng, J., Li, Y., Yuan, Z., Zhang, Y., Ji, J., & Zhang, Y. (2021). Neighbor-vote: Improving monocular 3d object detection through neighbor distance voting. In: Proceedings of the 29th ACM International Conference on Multimedia, pp 5239\u20135247","DOI":"10.1145\/3474085.3475641"},{"key":"2544_CR5","doi-asserted-by":"crossref","unstructured":"Dai, J., Qi, H., Xiong, Y., Li, Y., Zhang, G., Hu, H., & Wei, Y. (2017). Deformable convolutional networks., In: Proceedings of the IEEE international conference on computer vision, 764\u2013773.","DOI":"10.1109\/ICCV.2017.89"},{"key":"2544_CR6","doi-asserted-by":"publisher","first-page":"1201","DOI":"10.1609\/aaai.v35i2.16207","volume":"35","author":"J Deng","year":"2021","unstructured":"Deng, J., Shi, S., Li, P., Zhou, W., Zhang, Y., & Li, H. (2021). Voxel r-cnn: Towards high performance voxel-based 3d object detection. Proceedings of the AAAI conference on artificial intelligence, 35, 1201\u20131209.","journal-title":"Proceedings of the AAAI conference on artificial intelligence"},{"key":"2544_CR7","doi-asserted-by":"crossref","unstructured":"Ding, M., Huo, Y., Yi, H., Wang, Z., Shi, J., Lu, Z., Luo., P. (2020). Learning depth-guided convolutions for monocular 3d object detection. In: Proceedings of the IEEE\/CVF Conference on computer vision and pattern recognition workshops, pp 1000\u20131001","DOI":"10.1109\/CVPRW50498.2020.00508"},{"key":"2544_CR8","doi-asserted-by":"crossref","unstructured":"Fu, H., Gong, M., Wang, C., Batmanghelich, K., & Tao, D. (2018). Deep ordinal regression network for monocular depth estimation., In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2002\u20132011.","DOI":"10.1109\/CVPR.2018.00214"},{"key":"2544_CR9","doi-asserted-by":"crossref","unstructured":"Graham, B. (2015). Sparse 3d convolutional neural networks. In: Xie X, Jones MW, Tam GKL (eds) Proceedings of the British Machine Vision Conference 2015, pp 150.1\u2013150.9","DOI":"10.5244\/C.29.150"},{"key":"2544_CR10","doi-asserted-by":"crossref","unstructured":"Guizilini, V., Ambrus, R., Pillai, S., Raventos, A., & Gaidon, A. (2020). 3d packing for self-supervised monocular depth estimation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 2485\u20132494","DOI":"10.1109\/CVPR42600.2020.00256"},{"issue":"7","key":"2544_CR11","doi-asserted-by":"publisher","first-page":"6544","DOI":"10.1109\/LRA.2024.3401172","volume":"9","author":"C Han","year":"2024","unstructured":"Han, C., Yang, J., Sun, J., Ge, Z., Dong, R., Zhou, H., Mao, W., Peng, Y., & Zhang, X. (2024). Exploring recurrent long-term temporal fusion for multi-view 3d perception. IEEE Robotics Autom Lett, 9(7), 6544\u20136551.","journal-title":"IEEE Robotics Autom Lett"},{"key":"2544_CR12","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition., In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"2544_CR13","unstructured":"Huang, J., Huang, G. (2022). Bevdet4d: Exploit temporal cues in multi-camera 3d object detection. arXiv preprint arXiv:2203.17054"},{"key":"2544_CR14","unstructured":"Huang, J., Huang, G., Zhu, Z., Ye, Y., Du, D. (2021). Bevdet: High-performance multi-camera 3d object detection in bird-eye-view. arXiv preprint arXiv:2112.11790"},{"key":"2544_CR15","doi-asserted-by":"publisher","first-page":"2561","DOI":"10.1609\/aaai.v38i3.28033","volume":"38","author":"X Jiang","year":"2024","unstructured":"Jiang, X., Li, S., Liu, Y., Wang, S., Jia, F., Wang, T., Han, L., & Zhang, X. (2024). Far3d: Expanding the horizon for surround-view 3d object detection. Proceedings of the AAAI Conference on Artificial Intelligence, 38, 2561\u20132569.","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"2544_CR16","doi-asserted-by":"publisher","first-page":"1042","DOI":"10.1609\/aaai.v37i1.25185","volume":"37","author":"Y Jiang","year":"2023","unstructured":"Jiang, Y., Zhang, L., Miao, Z., Zhu, X., Gao, J., Hu, W., & Jiang, Y. G. (2023). Polarformer: Multi-camera 3d object detection with polar transformer. Proceedings of the AAAI conference on Artificial Intelligence, 37, 1042\u20131050.","journal-title":"Proceedings of the AAAI conference on Artificial Intelligence"},{"key":"2544_CR17","doi-asserted-by":"crossref","unstructured":"Kim, S., Kim, Y., Lee, I. J., & Kum, D. (2023). Predict to detect: Prediction-guided 3d object detection using sequential images. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 18057\u201318066","DOI":"10.1109\/ICCV51070.2023.01655"},{"key":"2544_CR18","doi-asserted-by":"crossref","unstructured":"Lee, Y., Hwang, J.w., Lee, S., Bae, Y., & Park, J. (2019). An energy and gpu-computation efficient backbone network for real-time object detection. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition workshops, pp 0\u20130","DOI":"10.1109\/CVPRW.2019.00103"},{"key":"2544_CR19","doi-asserted-by":"crossref","unstructured":"Lee, Y., Hwang, J.w., Lee, S., Bae, Y., & Park, J. (2019b). An energy and gpu-computation efficient backbone network for real-time object detection. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition workshops, pp 0\u20130","DOI":"10.1109\/CVPRW.2019.00103"},{"key":"2544_CR20","doi-asserted-by":"crossref","unstructured":"Li, Z., Wang, W., Li, H., Xie, E., Sima, C., Lu, T., Yu, Q., & Dai, J. (2024). Bevformer: learning bird\u2019s-eye-view representation from lidar-camera via spatiotemporal transformers. IEEE Transactions on Pattern Analysis and Machine Intelligence","DOI":"10.1109\/TPAMI.2024.3515454"},{"key":"2544_CR21","doi-asserted-by":"crossref","unstructured":"Li, Z., Yu, Z., Wang, W., Anandkumar, A., Lu, T., & Alvarez, J. M. (2023c). Fb-bev: Bev representation from forward-backward view transformations. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 6919\u20136928","DOI":"10.1109\/ICCV51070.2023.00637"},{"key":"2544_CR22","doi-asserted-by":"crossref","unstructured":"Li, P., Zhao, H., Liu, P., & Cao, F. (2020). Rtm3d: Real-time monocular 3d detection from object keypoints for autonomous driving. In: European Conference on Computer Vision, Springer, pp 644\u2013660","DOI":"10.1007\/978-3-030-58580-8_38"},{"key":"2544_CR23","doi-asserted-by":"publisher","first-page":"1486","DOI":"10.1609\/aaai.v37i2.25234","volume":"37","author":"Y Li","year":"2023","unstructured":"Li, Y., Bao, H., Ge, Z., Yang, J., Sun, J., & Li, Z. (2023). Bevstereo: Enhancing depth estimation in multi-view 3d object detection with temporal stereo. Proceedings of the AAAI Conference on Artificial Intelligence, 37, 1486\u20131494.","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"2544_CR24","first-page":"18442","volume":"35","author":"Y Li","year":"2022","unstructured":"Li, Y., Chen, Y., Qi, X., Li, Z., Sun, J., & Jia, J. (2022). Unifying voxel-based representation with transformer for 3d object detection. Advances in Neural Information Processing Systems, 35, 18442\u201318455.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2544_CR25","doi-asserted-by":"publisher","first-page":"1477","DOI":"10.1609\/aaai.v37i2.25233","volume":"37","author":"Y Li","year":"2023","unstructured":"Li, Y., Ge, Z., Yu, G., Yang, J., Wang, Z., Shi, Y., Sun, J., & Li, Z. (2023). Bevdepth: Acquisition of reliable depth for multi-view 3d object detection. Proceedings of the AAAI conference on artificial intelligence, 37, 1477\u20131485.","journal-title":"Proceedings of the AAAI conference on artificial intelligence"},{"key":"2544_CR26","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B., & Belongie, S. (2017). Feature pyramid networks for object detection. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2117\u20132125","DOI":"10.1109\/CVPR.2017.106"},{"key":"2544_CR27","unstructured":"Lin, X., Lin, T., Pei, Z., Huang, L., & Su, Z. (2022). Sparse4d: Multi-view 3d object detection with sparse spatial-temporal fusion. arXiv preprint arXiv:2211.10581"},{"key":"2544_CR28","unstructured":"Lin, X., Lin, T., Pei, Z., Huang, L., & Su, Z. (2023a). Sparse4d v2: Recurrent temporal fusion with sparse model. arXiv preprint arXiv:2305.14018"},{"key":"2544_CR29","unstructured":"Lin, X., Pei, Z., Lin, T., Huang, L., & Su, Z. (2023b). Sparse4d v3: Advancing end-to-end 3d detection and tracking. CoRR abs\/2311.11722"},{"key":"2544_CR30","doi-asserted-by":"crossref","unstructured":"Liu, F., Huang, T., Zhang, Q., Yao, H., Zhang, C., Wan, F., Ye, Q., & Zhou, Y. (2024). Ray denoising: Depth-aware hard negative sampling for multi-view 3d object detection. In: European Conference on Computer Vision, Springer, pp 200\u2013217","DOI":"10.1007\/978-3-031-72967-6_12"},{"key":"2544_CR31","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., & Guo, B. (2021). Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"2544_CR32","doi-asserted-by":"crossref","unstructured":"Liu, H., Teng, Y., Lu, T., Wang, H., & Wang, L. (2023a). Sparsebev: High-performance sparse 3d object detection from multi-camera videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 18580\u201318590","DOI":"10.1109\/ICCV51070.2023.01703"},{"key":"2544_CR33","doi-asserted-by":"crossref","unstructured":"Liu, Y., Wang, T., Zhang, X., & Sun, J. (2022). Petr: Position embedding transformation for multi-view 3d object detection. In: European conference on computer vision, Springer, pp 531\u2013548","DOI":"10.1007\/978-3-031-19812-0_31"},{"key":"2544_CR34","doi-asserted-by":"crossref","unstructured":"Liu, Z., Wu, Z., & T\u00f3th, R. (2020). Smoke: Single-stage monocular 3d object detection via keypoint estimation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition workshops, pp 996\u2013997","DOI":"10.1109\/CVPRW50498.2020.00506"},{"key":"2544_CR35","doi-asserted-by":"crossref","unstructured":"Liu, Y., Yan, J., Jia, F., Li, S., Gao, A., Wang, T., & Zhang, X. (2023b). Petrv2: A unified framework for 3d perception from multi-camera images. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 3262\u20133272","DOI":"10.1109\/ICCV51070.2023.00302"},{"key":"2544_CR36","unstructured":"Loshchilov, I., & Hutter, F. (2017). SGDR: stochastic gradient descent with warm restarts. In: International Conference on Learning Representations (ICLR)"},{"key":"2544_CR37","unstructured":"Loshchilov, I., & Hutter, F. (2019). Decoupled weight decay regularization. In: International Conference on Learning Representations (ICLR)"},{"key":"2544_CR38","doi-asserted-by":"crossref","unstructured":"Ma, X., Liu, S., Xia, Z., Zhang, H., Zeng, X., & Ouyang, W. (2020). Rethinking pseudo-lidar representation. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XIII 16, Springer, pp 311\u2013327","DOI":"10.1007\/978-3-030-58601-0_19"},{"key":"2544_CR39","doi-asserted-by":"crossref","unstructured":"Ma, X., Wang, Z., Li, H., Zhang, P., Ouyang, W., & Fan, X. (2019). Accurate monocular 3d object detection via color-embedded 3d reconstruction for autonomous driving. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 6851\u20136860","DOI":"10.1109\/ICCV.2019.00695"},{"issue":"8","key":"2544_CR40","doi-asserted-by":"publisher","first-page":"1909","DOI":"10.1007\/s11263-023-01790-1","volume":"131","author":"J Mao","year":"2023","unstructured":"Mao, J., Shi, S., Wang, X., & Li, H. (2023). 3d object detection for autonomous driving: A comprehensive survey. Int J Comput Vis, 131(8), 1909\u20131963.","journal-title":"Int J Comput Vis"},{"issue":"5","key":"2544_CR41","doi-asserted-by":"publisher","first-page":"3537","DOI":"10.1109\/TPAMI.2023.3346386","volume":"46","author":"X Ma","year":"2024","unstructured":"Ma, X., Ouyang, W., Simonelli, A., & Ricci, E. (2024). 3d object detection from images for autonomous driving: A survey. IEEE Trans Pattern Anal Mach Intell, 46(5), 3537\u20133556.","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"2544_CR42","doi-asserted-by":"crossref","unstructured":"Mousavian, A., Anguelov, D., Flynn, J., & Kosecka, J. (2017). 3d bounding box estimation using deep learning and geometry. In: Proceedings of the IEEE conference on Computer Vision and Pattern Recognition, pp 7074\u20137082","DOI":"10.1109\/CVPR.2017.597"},{"key":"2544_CR43","doi-asserted-by":"crossref","unstructured":"Park, D., Ambrus, R., Guizilini, V., Li, J., & Gaidon, A. (2021). Is pseudo-lidar needed for monocular 3d object detection? In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 3142\u20133152","DOI":"10.1109\/ICCV48922.2021.00313"},{"key":"2544_CR44","unstructured":"Park, J., Xu, C., Yang, S., Keutzer, K., Kitani, K.M., Tomizuka, M., & Zhan, W. (2023). Time will tell: New outlooks and A baseline for temporal multi-view 3d object detection. In: International Conference on Learning Representations, ICLR"},{"key":"2544_CR45","doi-asserted-by":"crossref","unstructured":"Philion, J., & Fidler, S. (2020) Lift, splat, shoot: Encoding images from arbitrary camera rigs by implicitly unprojecting to 3d. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XIV 16, Springer, (pp 194\u2013210).","DOI":"10.1007\/978-3-030-58568-6_12"},{"key":"2544_CR46","unstructured":"Qi, C.R., Su, H., Mo, K., & Guibas, L. J. (2017a). Pointnet: Deep learning on point sets for 3d classification and segmentation. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, (pp 652\u2013660)."},{"key":"2544_CR47","unstructured":"Qi, C. R., Yi, L., Su, H., & Guibas, L. J. (2017b). Pointnet++: Deep hierarchical feature learning on point sets in a metric space. Advances in neural information processing systems 30"},{"key":"2544_CR48","doi-asserted-by":"crossref","unstructured":"Reading, C., Harakeh, A., Chae, J., & Waslander, S. L. (2021) Categorical depth distribution network for monocular 3d object detection. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp 8555\u20138564).","DOI":"10.1109\/CVPR46437.2021.00845"},{"issue":"2","key":"2544_CR49","doi-asserted-by":"publisher","first-page":"531","DOI":"10.1007\/s11263-022-01710-9","volume":"131","author":"S Shi","year":"2023","unstructured":"Shi, S., Jiang, L., Deng, J., Wang, Z., Guo, C., Shi, J., Wang, X., & Li, H. (2023). Pv-rcnn++: Point-voxel feature set abstraction with local vector representation for 3d object detection. International Journal of Computer Vision, 131(2), 531\u2013551.","journal-title":"International Journal of Computer Vision"},{"key":"2544_CR50","doi-asserted-by":"crossref","unstructured":"Tian, Z., Shen, C., Chen, H., & He, T. (2019). Fcos: Fully convolutional one-stage object detection. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, (pp 9627\u20139636).","DOI":"10.1109\/ICCV.2019.00972"},{"key":"2544_CR51","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A. N., Kaiser, \u0141., & Polosukhin, I. (2017). Attention is all you need. Advances in neural information processing systems 30"},{"key":"2544_CR52","doi-asserted-by":"crossref","unstructured":"Wang, Y., Chao, W. L., Garg, D., Hariharan, B., Campbell, M., & Weinberger, K. Q. (2019). Pseudo-lidar from visual depth estimation: Bridging the gap in 3d object detection for autonomous driving. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp 8445\u20138453).","DOI":"10.1109\/CVPR.2019.00864"},{"key":"2544_CR53","unstructured":"Wang, Y., Guizilini, V. C., Zhang, T., Wang, Y., Zhao, H., & Solomon, J. (2022). Detr3d: 3d object detection from multi-view images via 3d-to-2d queries. In: Conference on Robot Learning, PMLR, pp 180\u2013191"},{"key":"2544_CR54","doi-asserted-by":"crossref","unstructured":"Wang, Z., Huang, Z., Fu, J., Wang, N., & Liu, S. (2023c). Object as query: Lifting any 2d object detector to 3d detection. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, (pp 3791\u20133800).","DOI":"10.1109\/ICCV51070.2023.00351"},{"key":"2544_CR55","doi-asserted-by":"crossref","unstructured":"Wang, S., Liu, Y., Wang, T., Li, Y., & Zhang, X. (2023a). Exploring object-centric temporal modeling for efficient multi-view 3d object detection. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, (pp 3621\u20133631).","DOI":"10.1109\/ICCV51070.2023.00335"},{"key":"2544_CR56","doi-asserted-by":"crossref","unstructured":"Wang, T., Zhu, X., Pang, J., & Lin, D. (2021). Fcos3d: Fully convolutional one-stage monocular 3d object detection. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, (pp 913\u2013922).","DOI":"10.1109\/ICCVW54120.2021.00107"},{"issue":"8","key":"2544_CR57","doi-asserted-by":"publisher","first-page":"2122","DOI":"10.1007\/s11263-023-01784-z","volume":"131","author":"Y Wang","year":"2023","unstructured":"Wang, Y., Mao, Q., Zhu, H., Deng, J., Zhang, Y., Ji, J., Li, H., & Zhang, Y. (2023). Multi-modal 3d object detection in autonomous driving: A survey. Int J Comput Vis, 131(8), 2122\u20132152.","journal-title":"Int J Comput Vis"},{"key":"2544_CR58","unstructured":"Wilson, B., Qi, W., Agarwal, T., Lambert, J., Singh, J., Khandelwal, S., Pan, B., Kumar, R., Hartnett, A., Pontes, J. K., Ramanan, D., Carr, P., & Hays, J. (2021). Argoverse 2: Next generation datasets for self-driving perception and forecasting. In: Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks 1, NeurIPS Datasets and Benchmarks 2021, December 2021, virtual"},{"key":"2544_CR59","doi-asserted-by":"crossref","unstructured":"Yin, T., Zhou, X., & Krahenbuhl, P. (2021). Center-based 3d object detection and tracking. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp 11784\u201311793).","DOI":"10.1109\/CVPR46437.2021.01161"},{"issue":"10","key":"2544_CR60","doi-asserted-by":"publisher","first-page":"2425","DOI":"10.1007\/s11263-022-01657-x","volume":"130","author":"\u00c9 Zablocki","year":"2022","unstructured":"Zablocki, \u00c9., Ben-Younes, H., P\u00e9rez, P., & Cord, M. (2022). Explainability of deep vision-based autonomous driving systems: Review and challenges. Int J Comput Vis, 130(10), 2425\u20132452.","journal-title":"Int J Comput Vis"},{"key":"2544_CR61","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Lu, J., & Zhou, J. (2021) Objects are different: Flexible monocular 3d object detection. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp 3289\u20133298).","DOI":"10.1109\/CVPR46437.2021.00330"},{"key":"2544_CR62","doi-asserted-by":"crossref","unstructured":"Zhang, R., Qiu, H., Wang, T., Guo, Z., Cui, Z., Qiao, Y., Li, H., & Gao, P. (2023). Monodetr: Depth-guided transformer for monocular 3d object detection. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, (pp 9155\u20139166).","DOI":"10.1109\/ICCV51070.2023.00840"},{"key":"2544_CR63","unstructured":"Zhang, Y., Zhu, Z., Zheng, W., Huang, J., Huang, G., Zhou, J., & Lu, J. (2022). Beverse: Unified perception and prediction in birds-eye-view for vision-centric autonomous driving. arXiv preprint arXiv:2205.09743"},{"key":"2544_CR64","unstructured":"Zhu, B., Jiang, Z., Zhou, X., Li, Z., & Yu, G. (2019). Class-balanced grouping and sampling for point cloud 3d object detection. arXiv preprint arXiv:1908.09492"},{"key":"2544_CR65","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., & Dai, J. (2021). Deformable DETR: deformable transformers for end-to-end object detection. In: International Conference on Learning Representations (ICLR)"},{"key":"2544_CR66","doi-asserted-by":"crossref","unstructured":"Zong, Z., Jiang, D., Song, G., Xue, Z., Su, J., Li, H., & Liu, Y. (2023). Temporal enhanced training of multi-view 3d object detector via historical object prediction. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, (pp 3781\u20133790).","DOI":"10.1109\/ICCV51070.2023.00350"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02544-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02544-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02544-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,12]],"date-time":"2025-11-12T06:29:24Z","timestamp":1762928964000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02544-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,25]]},"references-count":66,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2025,11]]}},"alternative-id":["2544"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02544-x","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,8,25]]},"assertion":[{"value":"22 July 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 July 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 August 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"There are no conflicts to declare.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}