{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,9]],"date-time":"2026-05-09T00:12:52Z","timestamp":1778285572536,"version":"3.51.4"},"reference-count":74,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100022963","name":"Key Research and Development Program of Zhejiang Province","doi-asserted-by":"publisher","award":["2024C01026"],"award-info":[{"award-number":["2024C01026"]}],"id":[{"id":"10.13039\/100022963","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100022963","name":"Key Research and Development Program of Zhejiang Province","doi-asserted-by":"publisher","award":["2024C01212"],"award-info":[{"award-number":["2024C01212"]}],"id":[{"id":"10.13039\/100022963","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62272140"],"award-info":[{"award-number":["62272140"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Engineering Applications of Artificial Intelligence"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.engappai.2026.114576","type":"journal-article","created":{"date-parts":[[2026,4,2]],"date-time":"2026-04-02T09:57:05Z","timestamp":1775123825000},"page":"114576","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"P1","title":["Target-aware proposal-level fusion for multi-modal three-dimensional detection"],"prefix":"10.1016","volume":"176","author":[{"given":"Zilong","family":"Zhao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Baofu","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuyu","family":"Yin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0241-0727","authenticated-orcid":false,"given":"Jilin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Youhuizi","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yan","family":"Zeng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Honghao","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.engappai.2026.114576_b1","series-title":"2025 7th International Congress on Human-Computer Interaction, Optimization and Robotic Applications","first-page":"1","article-title":"Machine learning-based orange quality classification: A hyperparameter optimization approach through puma optimizer","author":"Akbulut","year":"2025"},{"key":"10.1016\/j.engappai.2026.114576_b2","doi-asserted-by":"crossref","unstructured":"Bai, X., Hu, Z., Zhu, X., Huang, Q., Chen, Y., Fu, H., Tai, C.-L., 2022. Transfusion: Robust lidar-camera fusion for 3d object detection with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 1090\u20131099.","DOI":"10.1109\/CVPR52688.2022.00116"},{"key":"10.1016\/j.engappai.2026.114576_b3","doi-asserted-by":"crossref","unstructured":"Caesar, H., Bankiti, V., Lang, A.H., Vora, S., Liong, V.E., Xu, Q., Krishnan, A., Pan, Y., Baldan, G., Beijbom, O., 2020. nuscenes: A multimodal dataset for autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 11621\u201311631.","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"10.1016\/j.engappai.2026.114576_b4","series-title":"2023 IEEE\/CVF International Conference on Computer Vision","first-page":"18021","article-title":"ObjectFusion: Multi-modal 3D object detection with object-centric fusion","author":"Cai","year":"2023"},{"key":"10.1016\/j.engappai.2026.114576_b5","series-title":"BEVFusion4D: Learning lidar-camera fusion under bird\u2019s-eye-view via cross-modality guidance and temporal aggregation","author":"Cai","year":"2023"},{"key":"10.1016\/j.engappai.2026.114576_b6","series-title":"European Conference on Computer Vision","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.engappai.2026.114576_b7","series-title":"European Conference on Computer Vision","first-page":"628","article-title":"Deformable feature aggregation for dynamic multi-modal 3D object detection","author":"Chen","year":"2022"},{"key":"10.1016\/j.engappai.2026.114576_b8","doi-asserted-by":"crossref","unstructured":"Chen, Y., Li, Y., Zhang, X., Sun, J., Jia, J., 2022. Focal sparse convolutional networks for 3d object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 5428\u20135437.","DOI":"10.1109\/CVPR52688.2022.00535"},{"key":"10.1016\/j.engappai.2026.114576_b9","doi-asserted-by":"crossref","unstructured":"Chen, Y., Yu, Z., Chen, Y., Lan, S., Anandkumar, A., Jia, J., Alvarez, J.M., 2023. Focalformer3d: focusing on hard instance for 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 8394\u20138405.","DOI":"10.1109\/ICCV51070.2023.00771"},{"key":"10.1016\/j.engappai.2026.114576_b10","doi-asserted-by":"crossref","unstructured":"Chen, X., Zhang, T., Wang, Y., Wang, Y., Zhao, H., 2023. Futr3d: A unified sensor fusion framework for 3d detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 172\u2013181.","DOI":"10.1109\/CVPRW59228.2023.00022"},{"key":"10.1016\/j.engappai.2026.114576_b11","series-title":"MMDetection3D: OpenMMLab next-generation platform for general 3D object detection","author":"Contributors","year":"2020"},{"key":"10.1016\/j.engappai.2026.114576_b12","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"1201","article-title":"Voxel r-cnn: Towards high performance voxel-based 3d object detection","volume":"vol. 35","author":"Deng","year":"2021"},{"key":"10.1016\/j.engappai.2026.114576_b13","doi-asserted-by":"crossref","unstructured":"Duan, K., Bai, S., Xie, L., Qi, H., Huang, Q., Tian, Q., 2019. Centernet: Keypoint triplets for object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 6569\u20136578.","DOI":"10.1109\/ICCV.2019.00667"},{"key":"10.1016\/j.engappai.2026.114576_b14","doi-asserted-by":"crossref","first-page":"351","DOI":"10.52202\/068431-0026","article-title":"Fully sparse 3d object detection","volume":"35","author":"Fan","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114576_b15","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R., 2017. Mask R-CNN. In: 2017 IEEE International Conference on Computer Vision. ICCV, pp. 2980\u20132988.","DOI":"10.1109\/ICCV.2017.322"},{"key":"10.1016\/j.engappai.2026.114576_b16","series-title":"2016 IEEE Conference on Computer Vision and Pattern Recognition","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.engappai.2026.114576_b17","series-title":"EA-LSS: Edge-aware lift-splat-shot framework for 3D BEV object detection","author":"Hu","year":"2023"},{"key":"10.1016\/j.engappai.2026.114576_b18","series-title":"Bevdet: High-performance multi-camera 3d object detection in bird-eye-view","author":"Huang","year":"2021"},{"key":"10.1016\/j.engappai.2026.114576_b19","series-title":"European Conference on Computer Vision","first-page":"439","article-title":"Detecting as labeling: Rethinking lidar-camera fusion in 3d object detection","author":"Huang","year":"2025"},{"key":"10.1016\/j.engappai.2026.114576_b20","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"2561","article-title":"Far3d: Expanding the horizon for surround-view 3d object detection","volume":"vol. 38","author":"Jiang","year":"2024"},{"key":"10.1016\/j.engappai.2026.114576_b21","doi-asserted-by":"crossref","unstructured":"Jiao, Y., Jie, Z., Chen, S., Chen, J., Ma, L., Jiang, Y.-G., 2023a. MSMDFusion: Fusing LiDAR and Camera at Multiple Scales With Multi-Depth Seeds for 3D Object Detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 21643\u201321652.","DOI":"10.1109\/CVPR52729.2023.02073"},{"key":"10.1016\/j.engappai.2026.114576_b22","doi-asserted-by":"crossref","unstructured":"Jiao, Y., Jie, Z., Chen, S., Chen, J., Ma, L., Jiang, Y.-G., 2023b. Msmdfusion: Fusing lidar and camera at multiple scales with multi-depth seeds for 3d object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 21643\u201321652.","DOI":"10.1109\/CVPR52729.2023.02073"},{"key":"10.1016\/j.engappai.2026.114576_b23","doi-asserted-by":"crossref","unstructured":"Lang, A.H., Vora, S., Caesar, H., Zhou, L., Yang, J., Beijbom, O., 2019. Pointpillars: Fast encoders for object detection from point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 12697\u201312705.","DOI":"10.1109\/CVPR.2019.01298"},{"key":"10.1016\/j.engappai.2026.114576_b24","first-page":"18442","article-title":"Unifying voxel-based representation with transformer for 3d object detection","volume":"35","author":"Li","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"11","key":"10.1016\/j.engappai.2026.114576_b25","doi-asserted-by":"crossref","first-page":"7217","DOI":"10.1109\/TPAMI.2024.3392303","article-title":"Fully sparse fusion for 3D object detection","volume":"46","author":"Li","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.engappai.2026.114576_b26","doi-asserted-by":"crossref","unstructured":"Li, X., Ma, T., Hou, Y., Shi, B., Yang, Y., Liu, Y., Wu, X., Chen, Q., Li, Y., Qiao, Y., et al., 2023. Logonet: Towards accurate 3d object detection with local-to-global cross-modal fusion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 17524\u201317534.","DOI":"10.1109\/CVPR52729.2023.01681"},{"key":"10.1016\/j.engappai.2026.114576_b27","series-title":"European Conference on Computer Vision","first-page":"1","article-title":"Bevformer: Learning bird\u2019s-eye-view representation from multi-camera images via spatiotemporal transformers","author":"Li","year":"2022"},{"issue":"3","key":"10.1016\/j.engappai.2026.114576_b28","doi-asserted-by":"crossref","first-page":"2020","DOI":"10.1109\/TPAMI.2024.3515454","article-title":"BEVFormer: Learning bird\u2019s-eye-view representation from lidar-camera via spatiotemporal transformers","volume":"47","author":"Li","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.engappai.2026.114576_b29","doi-asserted-by":"crossref","first-page":"10421","DOI":"10.52202\/068431-0757","article-title":"Bevfusion: A simple and robust lidar-camera fusion framework","volume":"35","author":"Liang","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114576_b30","series-title":"2017 IEEE Conference on Computer Vision and Pattern Recognition","first-page":"936","article-title":"Feature pyramid networks for object detection","author":"Lin","year":"2017"},{"key":"10.1016\/j.engappai.2026.114576_b31","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P., 2017b. Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision. pp. 2980\u20132988.","DOI":"10.1109\/ICCV.2017.324"},{"key":"10.1016\/j.engappai.2026.114576_b32","series-title":"Sparse4D: Multi-view 3D object detection with sparse spatial-temporal fusion","author":"Lin","year":"2022"},{"key":"10.1016\/j.engappai.2026.114576_b33","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B., 2021. Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"10.1016\/j.engappai.2026.114576_b34","series-title":"2023 IEEE International Conference on Robotics and Automation","first-page":"2774","article-title":"Bevfusion: Multi-task multi-sensor fusion with unified bird\u2019s-eye view representation","author":"Liu","year":"2023"},{"key":"10.1016\/j.engappai.2026.114576_b35","doi-asserted-by":"crossref","unstructured":"Liu, H., Teng, Y., Lu, T., Wang, H., Wang, L., 2023. Sparsebev: High-performance sparse 3d object detection from multi-camera videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 18580\u201318590.","DOI":"10.1109\/ICCV51070.2023.01703"},{"key":"10.1016\/j.engappai.2026.114576_b36","first-page":"11653","article-title":"CBNet: A novel composite backbone network architecture for object detection","author":"Liu","year":"2020","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.engappai.2026.114576_b37","series-title":"European Conference on Computer Vision","first-page":"531","article-title":"Petr: Position embedding transformation for multi-view 3d object detection","author":"Liu","year":"2022"},{"key":"10.1016\/j.engappai.2026.114576_b38","series-title":"European Conference on Computer Vision","first-page":"531","article-title":"Petr: Position embedding transformation for multi-view 3d object detection","author":"Liu","year":"2022"},{"key":"10.1016\/j.engappai.2026.114576_b39","series-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2017"},{"key":"10.1016\/j.engappai.2026.114576_b40","doi-asserted-by":"crossref","unstructured":"Mao, J., Xue, Y., Niu, M., Bai, H., Feng, J., Liang, X., Xu, H., Xu, C., 2021. Voxel transformer for 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 3164\u20133173.","DOI":"10.1109\/ICCV48922.2021.00315"},{"key":"10.1016\/j.engappai.2026.114576_b41","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XIV 16","first-page":"194","article-title":"Lift, splat, shoot: Encoding images from arbitrary camera rigs by implicitly unprojecting to 3d","author":"Philion","year":"2020"},{"key":"10.1016\/j.engappai.2026.114576_b42","unstructured":"Qi, C.R., Yi, L., Su, H., Guibas, L.J., 2017. PointNet++: deep hierarchical feature learning on point sets in a metric space. In: Proceedings of the 31st International Conference on Neural Information Processing Systems. ISBN: 9781510860964, pp. 5105\u20135114."},{"key":"10.1016\/j.engappai.2026.114576_b43","doi-asserted-by":"crossref","first-page":"23192","DOI":"10.52202\/068431-1685","article-title":"Pointnext: Revisiting pointnet++ with improved training and scaling strategies","volume":"35","author":"Qian","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114576_b44","doi-asserted-by":"crossref","unstructured":"Reading, C., Harakeh, A., Chae, J., Waslander, S.L., 2021. Categorical depth distribution network for monocular 3d object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 8555\u20138564.","DOI":"10.1109\/CVPR46437.2021.00845"},{"key":"10.1016\/j.engappai.2026.114576_b45","doi-asserted-by":"crossref","unstructured":"Shi, S., Guo, C., Jiang, L., Wang, Z., Shi, J., Wang, X., Li, H., 2020. Pv-rcnn: Point-voxel feature set abstraction for 3d object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 10529\u201310538.","DOI":"10.1109\/CVPR42600.2020.01054"},{"issue":"2","key":"10.1016\/j.engappai.2026.114576_b46","doi-asserted-by":"crossref","first-page":"531","DOI":"10.1007\/s11263-022-01710-9","article-title":"PV-RCNN++: Point-voxel feature set abstraction with local vector representation for 3D object detection","volume":"131","author":"Shi","year":"2023","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.engappai.2026.114576_b47","series-title":"European Conference on Computer Vision","first-page":"35","article-title":"Pillarnet: Real-time and high-performance pillar-based 3d object detection","author":"Shi","year":"2022"},{"key":"10.1016\/j.engappai.2026.114576_b48","doi-asserted-by":"crossref","unstructured":"Shi, S., Wang, X., Li, H., 2019. Pointrcnn: 3d object proposal generation and detection from point cloud. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 770\u2013779.","DOI":"10.1109\/CVPR.2019.00086"},{"key":"10.1016\/j.engappai.2026.114576_b49","series-title":"2017 IEEE Winter Conference on Applications of Computer Vision","first-page":"464","article-title":"Cyclical learning rates for training neural networks","author":"Smith","year":"2017"},{"key":"10.1016\/j.engappai.2026.114576_b50","doi-asserted-by":"crossref","unstructured":"Vora, S., Lang, A.H., Helou, B., Beijbom, O., 2020. Pointpainting: Sequential fusion for 3d object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 4604\u20134612.","DOI":"10.1109\/CVPR42600.2020.00466"},{"key":"10.1016\/j.engappai.2026.114576_b51","series-title":"Conference on Robot Learning","first-page":"180","article-title":"Detr3d: 3d object detection from multi-view images via 3d-to-2d queries","author":"Wang","year":"2022"},{"key":"10.1016\/j.engappai.2026.114576_b52","series-title":"MV2dfusion: Leveraging modality-specific object semantics for multi-modal 3D detection","author":"Wang","year":"2024"},{"key":"10.1016\/j.engappai.2026.114576_b53","doi-asserted-by":"crossref","unstructured":"Wang, S., Liu, Y., Wang, T., Li, Y., Zhang, X., 2023a. Exploring object-centric temporal modeling for efficient multi-view 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 3621\u20133631.","DOI":"10.1109\/ICCV51070.2023.00335"},{"key":"10.1016\/j.engappai.2026.114576_b54","doi-asserted-by":"crossref","unstructured":"Wang, S., Liu, Y., Wang, T., Li, Y., Zhang, X., 2023b. Exploring object-centric temporal modeling for efficient multi-view 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 3621\u20133631.","DOI":"10.1109\/ICCV51070.2023.00335"},{"key":"10.1016\/j.engappai.2026.114576_b55","doi-asserted-by":"crossref","unstructured":"Wang, C., Ma, C., Zhu, M., Yang, X., 2021a. PointAugmenting: Cross-Modal Augmentation for 3D Object Detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 11794\u201311803.","DOI":"10.1109\/CVPR46437.2021.01162"},{"key":"10.1016\/j.engappai.2026.114576_b56","doi-asserted-by":"crossref","unstructured":"Wang, C., Ma, C., Zhu, M., Yang, X., 2021b. PointAugmenting: Cross-Modal Augmentation for 3D Object Detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 11794\u201311803.","DOI":"10.1109\/CVPR46437.2021.01162"},{"key":"10.1016\/j.engappai.2026.114576_b57","series-title":"European Conference on Computer Vision","first-page":"386","article-title":"Monocular 3d object detection with depth from motion","author":"Wang","year":"2022"},{"key":"10.1016\/j.engappai.2026.114576_b58","doi-asserted-by":"crossref","unstructured":"Wang, H., Shi, C., Shi, S., Lei, M., Wang, S., He, D., Schiele, B., Wang, L., 2023a. Dsvt: Dynamic sparse voxel transformer with rotated sets. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 13520\u201313529.","DOI":"10.1109\/CVPR52729.2023.01299"},{"key":"10.1016\/j.engappai.2026.114576_b59","doi-asserted-by":"crossref","unstructured":"Wang, H., Shi, C., Shi, S., Lei, M., Wang, S., He, D., Schiele, B., Wang, L., 2023b. Dsvt: Dynamic sparse voxel transformer with rotated sets. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 13520\u201313529.","DOI":"10.1109\/CVPR52729.2023.01299"},{"key":"10.1016\/j.engappai.2026.114576_b60","doi-asserted-by":"crossref","unstructured":"Wang, K., Zhang, X., 2024. Deformable Shape-aware Point Generation for 3D Object Detection. In: Proceedings of the Asian Conference on Computer Vision. ACCV, pp. 2699\u20132715.","DOI":"10.1007\/978-981-96-0972-7_3"},{"key":"10.1016\/j.engappai.2026.114576_b61","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, J., Lee, J.-Y., Kweon, I.S., 2018. Cbam: Convolutional block attention module. In: Proceedings of the European Conference on Computer Vision. ECCV, pp. 3\u201319.","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"10.1016\/j.engappai.2026.114576_b62","doi-asserted-by":"crossref","unstructured":"Xie, Y., Xu, C., Rakotosaona, M.-J., Rim, P., Tombari, F., Keutzer, K., Tomizuka, M., Zhan, W., 2023. Sparsefusion: Fusing multi-modal sparse representations for multi-sensor 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 17591\u201317602.","DOI":"10.1109\/ICCV51070.2023.01613"},{"key":"10.1016\/j.engappai.2026.114576_b63","doi-asserted-by":"crossref","unstructured":"Xu, J., Peng, L., Cheng, H., Li, H., Qian, W., Li, K., Wang, W., Cai, D., 2023. Mononerd: Nerf-like representations for monocular 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 6814\u20136824.","DOI":"10.1109\/ICCV51070.2023.00627"},{"key":"10.1016\/j.engappai.2026.114576_b64","doi-asserted-by":"crossref","unstructured":"Yan, J., Liu, Y., Sun, J., Jia, F., Li, S., Wang, T., Zhang, X., 2023a. Cross modal transformer: Towards fast and robust 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 18268\u201318278.","DOI":"10.1109\/ICCV51070.2023.01675"},{"key":"10.1016\/j.engappai.2026.114576_b65","series-title":"Cross modal transformer via coordinates encoding for 3d object dectection","first-page":"4","author":"Yan","year":"2023"},{"issue":"10","key":"10.1016\/j.engappai.2026.114576_b66","doi-asserted-by":"crossref","first-page":"3337","DOI":"10.3390\/s18103337","article-title":"Second: Sparsely embedded convolutional detection","volume":"18","author":"Yan","year":"2018","journal-title":"Sensors"},{"key":"10.1016\/j.engappai.2026.114576_b67","first-page":"1992","article-title":"Deepinteraction: 3d object detection via modality interaction","volume":"35","author":"Yang","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114576_b68","doi-asserted-by":"crossref","unstructured":"Yang, Z., Sun, Y., Liu, S., Jia, J., 2020. 3dssd: Point-based 3d single stage object detector. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 11040\u201311048.","DOI":"10.1109\/CVPR42600.2020.01105"},{"key":"10.1016\/j.engappai.2026.114576_b69","doi-asserted-by":"crossref","unstructured":"Yin, T., Zhou, X., Krahenbuhl, P., 2021a. Center-based 3d object detection and tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 11784\u201311793.","DOI":"10.1109\/CVPR46437.2021.01161"},{"key":"10.1016\/j.engappai.2026.114576_b70","first-page":"16494","article-title":"Multimodal virtual point 3d detection","volume":"34","author":"Yin","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114576_b71","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Tuzel, O., 2018. VoxelNet: End-to-End Learning for Point Cloud Based 3D Object Detection. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 4490\u20134499.","DOI":"10.1109\/CVPR.2018.00472"},{"key":"10.1016\/j.engappai.2026.114576_b72","series-title":"European Conference on Computer Vision","first-page":"496","article-title":"Centerformer: Center-based transformer for 3d object detection","author":"Zhou","year":"2022"},{"key":"10.1016\/j.engappai.2026.114576_b73","series-title":"Class-balanced grouping and sampling for point cloud 3d object detection","author":"Zhu","year":"2019"},{"key":"10.1016\/j.engappai.2026.114576_b74","series-title":"Deformable detr: Deformable transformers for end-to-end object detection","author":"Zhu","year":"2020"}],"container-title":["Engineering Applications of Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626008572?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626008572?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,8]],"date-time":"2026-05-08T23:30:27Z","timestamp":1778283027000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0952197626008572"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":74,"alternative-id":["S0952197626008572"],"URL":"https:\/\/doi.org\/10.1016\/j.engappai.2026.114576","relation":{},"ISSN":["0952-1976"],"issn-type":[{"value":"0952-1976","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Target-aware proposal-level fusion for multi-modal three-dimensional detection","name":"articletitle","label":"Article Title"},{"value":"Engineering Applications of Artificial Intelligence","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.engappai.2026.114576","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"114576"}}