{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T23:50:54Z","timestamp":1782863454459,"version":"3.54.5"},"reference-count":72,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T00:00:00Z","timestamp":1779235200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T00:00:00Z","timestamp":1779235200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U21B2028"],"award-info":[{"award-number":["U21B2028"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Machine Vision and Applications"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s00138-026-01834-9","type":"journal-article","created":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T17:28:41Z","timestamp":1779298121000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Multi-modal Gaussian splatting for 3D object detection"],"prefix":"10.1007","volume":"37","author":[{"given":"Shuai","family":"Guo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qiyue","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhenpeng","family":"Yin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiajun","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingyi","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shangjing","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongyang","family":"Bai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,20]]},"reference":[{"key":"1834_CR1","unstructured":"Wang, Y., Ye, J.: An overview of 3d object detection, (2020). (arXiv preprint)"},{"key":"1834_CR2","doi-asserted-by":"publisher","first-page":"5291","DOI":"10.1109\/TMM.2022.3189778","volume":"25","author":"H Zhu","year":"2022","unstructured":"Zhu, H., Deng, J., Zhang, Y., Ji, J., Mao, Q., Li, H., Zhang, Y.: Vpfnet: Improving 3d object detection with virtual point based lidar and stereo data fusion. IEEE Trans. Multimedia 25, 5291\u20135304 (2022)","journal-title":"IEEE Trans. Multimedia"},{"key":"1834_CR3","doi-asserted-by":"publisher","first-page":"3388","DOI":"10.1109\/TMM.2020.3025166","volume":"23","author":"W Zhou","year":"2020","unstructured":"Zhou, W., Wu, J., Lei, J., Hwang, J.-N., Yu, L.: Salient object detection in stereoscopic 3d images using a deep convolutional residual autoencoder. IEEE Trans. Multimedia 23, 3388\u20133399 (2020)","journal-title":"IEEE Trans. Multimedia"},{"issue":"8","key":"1834_CR4","doi-asserted-by":"publisher","first-page":"1909","DOI":"10.1007\/s11263-023-01790-1","volume":"131","author":"J Mao","year":"2023","unstructured":"Mao, J., Shi, S., Wang, X., Li, H.: 3d object detection for autonomous driving: A comprehensive survey. Int. J. Comput. Vision 131(8), 1909\u20131963 (2023)","journal-title":"Int. J. Comput. Vision"},{"issue":"1","key":"1834_CR5","doi-asserted-by":"publisher","first-page":"1019","DOI":"10.1109\/TVCG.2021.3114853","volume":"28","author":"S Jamonnak","year":"2021","unstructured":"Jamonnak, S., Zhao, Y., Huang, X., Amiruzzaman, M.: Geo-context aware study of vision-based autonomous driving models and spatial video data. IEEE Trans. Visual Comput. Graphics 28(1), 1019\u20131029 (2021)","journal-title":"IEEE Trans. Visual Comput. Graphics"},{"key":"1834_CR6","doi-asserted-by":"crossref","unstructured":"Xu, G., Khan, A.S., Moshayedi, A.J., Zhang, X., Shuxin, Y.: The object detection, perspective and obstacles in robotic: a review. EAI Endorsed Transactions on AI and Robotics 1(1) (2022)","DOI":"10.4108\/airo.v1i1.2709"},{"key":"1834_CR7","doi-asserted-by":"crossref","unstructured":"Gao, H., Sun, Y., Xiao, J., Fang, D., Xu, Y., Wei, W.: Toward effective 3d object detection via multimodal fusion to automatic driving for industrial cyber-physical systems. IEEE Transactions on Industrial Cyber-Physical Systems , (2024)","DOI":"10.1109\/TICPS.2024.3427060"},{"key":"1834_CR8","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2024.124945","volume":"256","author":"Y Tang","year":"2024","unstructured":"Tang, Y., He, H., Wang, Y., Wu, J.: Towards efficient multi-modal 3d object detection: Homogeneous sparse fuse network. Expert Syst. Appl. 256, 124945 (2024)","journal-title":"Expert Syst. Appl."},{"issue":"2","key":"1834_CR9","doi-asserted-by":"publisher","first-page":"1391","DOI":"10.1109\/JSEN.2021.3127626","volume":"22","author":"Z Zhang","year":"2021","unstructured":"Zhang, Z., Liang, Z., Zhang, M., Zhao, X., Li, H., Yang, M., Tan, W., Pu, S.: Rangelvdet: Boosting 3d object detection in lidar with range image and rgb image. IEEE Sens. J. 22(2), 1391\u20131403 (2021)","journal-title":"IEEE Sens. J."},{"key":"1834_CR10","doi-asserted-by":"crossref","unstructured":"Paigwar, A., Sierra-Gonzalez, D., Erkent, \u00d6., Laugier, C.: Frustum-pointpillars: A multi-stage approach for 3d object detection using rgb camera and lidar. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2926\u20132933 (2021)","DOI":"10.1109\/ICCVW54120.2021.00327"},{"key":"1834_CR11","doi-asserted-by":"crossref","unstructured":"Chen, X., Kundu, K., Zhang, Z., Ma, H., Fidler, S., Urtasun, R.: Monocular 3d object detection for autonomous driving. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2147\u20132156. (2016)","DOI":"10.1109\/CVPR.2016.236"},{"key":"1834_CR12","doi-asserted-by":"crossref","unstructured":"Qi, C.R., Liu, W., Wu, C., Su, H., Guibas, L.J.: Frustum pointnets for 3d object detection from rgb-d data. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 918\u2013927. (2018)","DOI":"10.1109\/CVPR.2018.00102"},{"key":"1834_CR13","doi-asserted-by":"crossref","unstructured":"Vora, S., Lang, A.H., Helou, B., Beijbom, O.: Pointpainting: Sequential fusion for 3d object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4604\u20134612. (2020)","DOI":"10.1109\/CVPR42600.2020.00466"},{"issue":"9","key":"1834_CR14","doi-asserted-by":"publisher","first-page":"2137","DOI":"10.1109\/TVCG.2016.2601915","volume":"23","author":"J Wang","year":"2016","unstructured":"Wang, J., Xu, K.: Shape detection from raw lidar data with subspace modeling. IEEE Trans. Visual Comput. Graphics 23(9), 2137\u20132150 (2016)","journal-title":"IEEE Trans. Visual Comput. Graphics"},{"key":"1834_CR15","doi-asserted-by":"crossref","unstructured":"Chen, X., Ma, H., Wan, J., Li, B., Xia, T.: Multi-view 3d object detection network for autonomous driving. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1907\u20131915. (2017)","DOI":"10.1109\/CVPR.2017.691"},{"key":"1834_CR16","first-page":"16494","volume":"34","author":"T Yin","year":"2021","unstructured":"Yin, T., Zhou, X., Kr\u00e4henb\u00fchl, P.: Multimodal virtual point 3d detection. Adv. Neural. Inf. Process. Syst. 34, 16494\u201316507 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1834_CR17","doi-asserted-by":"crossref","unstructured":"Pang, S., Morris, D., Radha, H.: Clocs: Camera-lidar object candidates fusion for 3d object detection. In: 2020 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 10386\u201310393. IEEE (2020)","DOI":"10.1109\/IROS45743.2020.9341791"},{"key":"1834_CR18","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Tuzel, O.: Voxelnet: End-to-end learning for point cloud based 3d object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4490\u20134499. (2018)","DOI":"10.1109\/CVPR.2018.00472"},{"key":"1834_CR19","doi-asserted-by":"crossref","unstructured":"Brazil, G., Liu, X.: M3d-rpn: Monocular 3d region proposal network for object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9287\u20139296. (2019)","DOI":"10.1109\/ICCV.2019.00938"},{"key":"1834_CR20","doi-asserted-by":"crossref","unstructured":"Fu, H., Gong, M., Wang, C., Batmanghelich, K., Tao, D.: Deep ordinal regression network for monocular depth estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2002\u20132011. (2018)","DOI":"10.1109\/CVPR.2018.00214"},{"key":"1834_CR21","doi-asserted-by":"crossref","unstructured":"Ku, J., Pon, A.D., Waslander, S.L.: Monocular 3d object detection leveraging accurate proposals and shape reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11867\u201311876. (2019)","DOI":"10.1109\/CVPR.2019.01214"},{"key":"1834_CR22","doi-asserted-by":"crossref","unstructured":"Li, P., Zhao, H., Liu, P., Cao, F.: Rtm3d: Real-time monocular 3d detection from object keypoints for autonomous driving. In: European Conference on Computer Vision, pp. 644\u2013660 (2020). Springer","DOI":"10.1007\/978-3-030-58580-8_38"},{"key":"1834_CR23","doi-asserted-by":"crossref","unstructured":"Brazil, G., Pons-Moll, G., Liu, X., Schiele, B.: Kinematic 3d object detection in monocular video. In: Computer Vision-ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXIII 16, pp. 135\u2013152 (2020). Springer","DOI":"10.1007\/978-3-030-58592-1_9"},{"key":"1834_CR24","doi-asserted-by":"crossref","unstructured":"Heylen, J., De Wolf, M., Dawagne, B., Proesmans, M., Van Gool, L., Abbeloos, W., Abdelkawy, H., Reino, D.O.: Monocinis: Camera independent monocular 3d object detection using instance segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 923\u2013934. (2021)","DOI":"10.1109\/ICCVW54120.2021.00108"},{"key":"1834_CR25","doi-asserted-by":"crossref","unstructured":"Yang, Z., Sun, Y., Liu, S., Shen, X., Jia, J.: Std: Sparse-to-dense 3d object detector for point cloud. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1951\u20131960. (2019)","DOI":"10.1109\/ICCV.2019.00204"},{"key":"1834_CR26","unstructured":"Liu, Z., Tang, H., Lin, Y., Han, S.: Point-voxel cnn for efficient 3d deep learning. Adv. Neural. Inf. Process. Syst. 32, (2019)"},{"key":"1834_CR27","doi-asserted-by":"crossref","unstructured":"Meyer, G.P., Laddha, A., Kee, E., Vallespi-Gonzalez, C., Wellington, C.K.: Lasernet: An efficient probabilistic 3d object detector for autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12677\u201312686. (2019)","DOI":"10.1109\/CVPR.2019.01296"},{"key":"1834_CR28","doi-asserted-by":"crossref","unstructured":"Wang, C., Ma, C., Zhu, M., Yang, X.: Pointaugmenting: Cross-modal augmentation for 3d object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11794\u201311803. (2021)","DOI":"10.1109\/CVPR46437.2021.01162"},{"key":"1834_CR29","doi-asserted-by":"crossref","unstructured":"Pang, S., Morris, D., Radha, H.: Fast-clocs: Fast camera-lidar object candidates fusion for 3d object detection. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 187\u2013196. (2022)","DOI":"10.1109\/WACV51458.2022.00380"},{"key":"1834_CR30","doi-asserted-by":"crossref","unstructured":"Chen, X., Zhang, T., Wang, Y., Wang, Y., Zhao, H.: Futr3d: A unified sensor fusion framework for 3d detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 172\u2013181 (2023)","DOI":"10.1109\/CVPRW59228.2023.00022"},{"key":"1834_CR31","doi-asserted-by":"crossref","unstructured":"Chen, Y., Li, Y., Zhang, X., Sun, J., Jia, J.: Focal sparse convolutional networks for 3d object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5428\u20135437. (2022)","DOI":"10.1109\/CVPR52688.2022.00535"},{"key":"1834_CR32","doi-asserted-by":"crossref","unstructured":"Chen, Z., Li, Z., Zhang, S., Fang, L., Jiang, Q., Zhao, F., Zhou, B., Zhao, H.: Autoalign: Pixel-instance feature aggregation for multi-modal 3d object detection. arXiv preprint arXiv:2201.06493 (2022)","DOI":"10.24963\/ijcai.2022\/116"},{"key":"1834_CR33","doi-asserted-by":"crossref","unstructured":"Li, Y., Yu, A.W., Meng, T., Caine, B., Ngiam, J., Peng, D., Shen, J., Lu, Y., Zhou, D., Le, Q.V.: Deepfusion: Lidar-camera deep fusion for multi-modal 3d object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17182\u201317191. (2022)","DOI":"10.1109\/CVPR52688.2022.01667"},{"key":"1834_CR34","doi-asserted-by":"crossref","unstructured":"Wu, X., Peng, L., Yang, H., Xie, L., Huang, C., Deng, C., Liu, H., Cai, D.: Sparse fuse dense: Towards high quality 3d detection with depth completion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5418\u20135427. (2022)","DOI":"10.1109\/CVPR52688.2022.00534"},{"key":"1834_CR35","doi-asserted-by":"crossref","unstructured":"Bai, X., Hu, Z., Zhu, X., Huang, Q., Chen, Y., Fu, H., Tai, C.-L.: Transfusion: Robust lidar-camera fusion for 3d object detection with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1090\u20131099 (2022)","DOI":"10.1109\/CVPR52688.2022.00116"},{"issue":"1","key":"1834_CR36","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1145\/3503250","volume":"65","author":"B Mildenhall","year":"2021","unstructured":"Mildenhall, B., Srinivasan, P.P., Tancik, M., Barron, J.T., Ramamoorthi, R., Ng, R.: Nerf: Representing scenes as neural radiance fields for view synthesis. Commun. ACM 65(1), 99\u2013106 (2021)","journal-title":"Commun. ACM"},{"key":"1834_CR37","doi-asserted-by":"crossref","unstructured":"Hu, B., Huang, J., Liu, Y., Tai, Y.-W., Tang, C.-K.: Nerf-rpn: A general framework for object detection in nerfs. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23528\u201323538 (2023)","DOI":"10.1109\/CVPR52729.2023.02253"},{"key":"1834_CR38","doi-asserted-by":"crossref","unstructured":"Xu, C., Wu, B., Hou, J., Tsai, S., Li, R., Wang, J., Zhan, W., He, Z., Vajda, P., Keutzer, K.: Nerf-det: Learning geometry-aware volumetric representation for multi-view 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 23320\u201323330. (2023)","DOI":"10.1109\/ICCV51070.2023.02131"},{"issue":"4","key":"1834_CR39","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1145\/3592433","volume":"42","author":"B Kerbl","year":"2023","unstructured":"Kerbl, B., Kopanas, G., Leimk\u00fchler, T., Drettakis, G.: 3d gaussian splatting for real-time radiance field rendering. ACM Trans. Graph. 42(4), 139\u20131 (2023)","journal-title":"ACM Trans. Graph."},{"key":"1834_CR40","unstructured":"Cao, Y., Jv, Y., Xu, D.: 3dgs-det: Empower 3d gaussian splatting with boundary guidance and box-focused sampling for 3d object detection. arXiv preprint arXiv:2410.01647 (2024)"},{"key":"1834_CR41","doi-asserted-by":"crossref","unstructured":"Zhu, Z., Fan, Z., Jiang, Y., Wang, Z.: FSGS: Real-Time Few-Shot View Synthesis using Gaussian Splatting, (2023)","DOI":"10.1007\/978-3-031-72933-1_9"},{"key":"1834_CR42","unstructured":"Mehta, S., Rastegari, M.: Mobilevit: light-weight, general-purpose, and mobile-friendly vision transformer, (2021). (arXiv preprint)"},{"key":"1834_CR43","doi-asserted-by":"crossref","unstructured":"He, X., Deng, K., Wang, X., Li, Y., Zhang, Y., Wang, M.: Lightgcn: Simplifying and powering graph convolution network for recommendation. In: Proceedings of the 43rd International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 639\u2013648 (2020)","DOI":"10.1145\/3397271.3401063"},{"key":"1834_CR44","unstructured":"Wang, S., Li, B.Z., Khabsa, M., Fang, H., Ma, H.: Linformer: Self-attention with linear complexity, (2020). (arXiv preprint)"},{"key":"1834_CR45","doi-asserted-by":"crossref","unstructured":"Chen, A., Xu, Z., Geiger, A., Yu, J., Su, H.: European Conference on Computer Vision, pp. 333\u2013350. Springer (2022). (Tensorf: Tensorial radiance fields)","DOI":"10.1007\/978-3-031-19824-3_20"},{"key":"1834_CR46","doi-asserted-by":"publisher","first-page":"23192","DOI":"10.52202\/068431-1685","volume":"35","author":"G Qian","year":"2022","unstructured":"Qian, G., Li, Y., Peng, H., Mai, J., Hammoud, H., Elhoseiny, M., Ghanem, B.: Pointnext: Revisiting pointnet++ with improved training and scaling strategies. Adv. Neural. Inf. Process. Syst. 35, 23192\u201323204 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1834_CR47","doi-asserted-by":"crossref","unstructured":"Zhu, S., Wang, X., Lai, S., Chen, Y., Zhai, W., Quan, D., Qi, Y., Lv, L.: Multi-graph aggregated graph neural network for heterogeneous graph representation learning. Int. J. Mach. Learn. Cybern. , 1\u201316 (2024)","DOI":"10.1007\/s13042-024-02294-1"},{"issue":"1","key":"1834_CR48","doi-asserted-by":"publisher","first-page":"24","DOI":"10.1007\/s11063-024-11496-1","volume":"56","author":"E Jiawei","year":"2024","unstructured":"Jiawei, E., Zhang, Y., Yang, S., Wang, H., Xia, X., Xu, X.: Graphsage++: Weighted multi-scale gnn for graph representation learning. Neural Process. Lett. 56(1), 24 (2024)","journal-title":"Neural Process. Lett."},{"key":"1834_CR49","doi-asserted-by":"crossref","unstructured":"Geiger, A., Lenz, P., Urtasun, R.: Are we ready for autonomous driving? the kitti vision benchmark suite. In: Conference on Computer Vision and Pattern Recognition (CVPR), (2012)","DOI":"10.1109\/CVPR.2012.6248074"},{"key":"1834_CR50","doi-asserted-by":"crossref","unstructured":"Dai, A., Chang, A.X., Savva, M., Halber, M., Funkhouser, T., Nie\u00dfner, M.: Scannet: Richly-annotated 3d reconstructions of indoor scenes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5828\u20135839. (2017)","DOI":"10.1109\/CVPR.2017.261"},{"key":"1834_CR51","doi-asserted-by":"crossref","unstructured":"Song, S., Lichtenberg, S.P., Xiao, J.: Sun rgb-d: A rgb-d scene understanding benchmark suite. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 567\u2013576. (2015)","DOI":"10.1109\/CVPR.2015.7298655"},{"key":"1834_CR52","doi-asserted-by":"crossref","unstructured":"Xu, D., Anguelov, D., Jain, A.: Pointfusion: Deep sensor fusion for 3d bounding box estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 244\u2013253. (2018)","DOI":"10.1109\/CVPR.2018.00033"},{"key":"1834_CR53","doi-asserted-by":"crossref","unstructured":"Zhao, X., Liu, Z., Hu, R., Huang, K.: 3d object detection using scale invariant and feature reweighting networks. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, pp. 9267\u20139274 (2019)","DOI":"10.1609\/aaai.v33i01.33019267"},{"key":"1834_CR54","doi-asserted-by":"crossref","unstructured":"Huang, T., Liu, Z., Chen, X., Bai, X.: Epnet: Enhancing point features with image semantics for 3d object detection. In: Computer Vision-ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XV 16, pp. 35\u201352 (2020). Springer","DOI":"10.1007\/978-3-030-58555-6_3"},{"key":"1834_CR55","doi-asserted-by":"crossref","unstructured":"Qi, C.R., Litany, O., He, K., Guibas, L.J.: Deep hough voting for 3d object detection in point clouds. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9277\u20139286. (2019)","DOI":"10.1109\/ICCV.2019.00937"},{"key":"1834_CR56","doi-asserted-by":"crossref","unstructured":"Rukhovich, D., Vorontsova, A., Konushin, A.: Fcaf3d: Fully convolutional anchor-free 3d object detection. In: European Conference on Computer Vision, pp. 477\u2013493. Springer (2022)","DOI":"10.1007\/978-3-031-20080-9_28"},{"key":"1834_CR57","doi-asserted-by":"publisher","first-page":"29975","DOI":"10.52202\/068431-2173","volume":"35","author":"H Wang","year":"2022","unstructured":"Wang, H., Ding, L., Dong, S., Shi, S., Li, A., Li, J., Li, Z., Wang, L.: Cagroup3d: Class-aware grouping for 3d object detection on point clouds. Adv. Neural. Inf. Process. Syst. 35, 29975\u201329988 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1834_CR58","doi-asserted-by":"crossref","unstructured":"Tu, T., Chuang, S.-P., Liu, Y.-L., Sun, C., Zhang, K., Roy, D., Kuo, C.-H., Sun, M.: Imgeonet: Image-induced geometry-aware voxel representation for multi-view 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6996\u20137007 (2023)","DOI":"10.1109\/ICCV51070.2023.00644"},{"key":"1834_CR59","doi-asserted-by":"crossref","unstructured":"Shen, G., Huang, J., Hu, Z., Wang, B.: Cn-rma: Combined network with ray marching aggregation for 3d indoor object detection from multi-view images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 21326\u201321335. (2024)","DOI":"10.1109\/CVPR52733.2024.02015"},{"key":"1834_CR60","doi-asserted-by":"crossref","unstructured":"Rukhovich, D., Vorontsova, A., Konushin, A.: Imvoxelnet: Image to voxels projection for monocular and multi-view general-purpose 3d object detection. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 2397\u20132406. (2022)","DOI":"10.1109\/WACV51458.2022.00133"},{"key":"1834_CR61","doi-asserted-by":"crossref","unstructured":"Cheng, B., Sheng, L., Shi, S., Yang, M., Xu, D.: Back-tracing representative points for voting-based 3d object detection in point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8963\u20138972. (2021)","DOI":"10.1109\/CVPR46437.2021.00885"},{"key":"1834_CR62","doi-asserted-by":"crossref","unstructured":"Misra, I., Girdhar, R., Joulin, A.: An end-to-end transformer model for 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2906\u20132917. (2021)","DOI":"10.1109\/ICCV48922.2021.00290"},{"key":"1834_CR63","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Sun, B., Yang, H., Huang, Q.: H3dnet: 3d object detection using hybrid geometric primitives. In: Computer Vision-ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XII 16, pp. 311\u2013329 (2020). Springer","DOI":"10.1007\/978-3-030-58610-2_19"},{"key":"1834_CR64","doi-asserted-by":"crossref","unstructured":"Liu, Z., Zhang, Z., Cao, Y., Hu, H., Tong, X.: Group-free 3d object detection via transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2949\u20132958. (2021)","DOI":"10.1109\/ICCV48922.2021.00294"},{"key":"1834_CR65","doi-asserted-by":"crossref","unstructured":"Wang, H., Shi, S., Yang, Z., Fang, R., Qian, Q., Li, H., Schiele, B., Wang, L.: Rbgnet: Ray-based grouping for 3d object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1110\u20131119 (2022)","DOI":"10.1109\/CVPR52688.2022.00118"},{"key":"1834_CR66","doi-asserted-by":"crossref","unstructured":"Zheng, Y., Duan, Y., Lu, J., Zhou, J., Tian, Q.: Hyperdet3d: Learning a scene-conditioned 3d object detector. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5585\u20135594. (2022)","DOI":"10.1109\/CVPR52688.2022.00550"},{"key":"1834_CR67","doi-asserted-by":"crossref","unstructured":"Rukhovich, D., Vorontsova, A., Konushin, A.: Tr3d: Towards real-time indoor 3d object detection. In: 2023 IEEE International Conference on Image Processing (ICIP), pp. 281\u2013285. IEEE (2023)","DOI":"10.1109\/ICIP49359.2023.10222644"},{"key":"1834_CR68","unstructured":"Paszke, A., Gross, S., Massa, F., Lerer, A., Bradbury, J., Chanan, G., Killeen, T., Lin, Z., Gimelshein, N., Antiga, L.: Pytorch: An imperative style, high-performance deep learning library. Adv. Neural. Inf. Process. Syst. 32, (2019)"},{"issue":"7","key":"1834_CR69","doi-asserted-by":"publisher","first-page":"1240","DOI":"10.3390\/rs17071240","volume":"17","author":"R Qiao","year":"2025","unstructured":"Qiao, R., Yuan, H., Guan, Z., Zhang, W.: Mdfusion: Multi-dimension semantic-spatial feature fusion for lidar-camera 3d object detection. Remote Sensing 17(7), 1240 (2025)","journal-title":"Remote Sensing"},{"key":"1834_CR70","doi-asserted-by":"crossref","unstructured":"Wang, Z., Huang, Z., Gao, Y., Wang, N., Liu, S.: Mv2dfusion: Leveraging modality-specific object semantics for multi-modal 3d detection. IEEE Transactions on Pattern Analysis and Machine Intelligence (2025)","DOI":"10.1109\/TPAMI.2025.3609348"},{"key":"1834_CR71","doi-asserted-by":"crossref","unstructured":"Zheng, C., Yan, X., Zhang, H., Wang, B., Cheng, S., Cui, S., Li, Z.: Beyond 3d siamese tracking: A motion-centric paradigm for 3d single object tracking in point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8111\u20138120. (2022)","DOI":"10.1109\/CVPR52688.2022.00794"},{"key":"1834_CR72","doi-asserted-by":"crossref","unstructured":"He, C., Li, R., Li, S., Zhang, L.: Voxel set transformer: A set-to-set approach to 3d object detection from point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8417\u20138427 (2022)","DOI":"10.1109\/CVPR52688.2022.00823"}],"container-title":["Machine Vision and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-026-01834-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00138-026-01834-9","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-026-01834-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T22:56:03Z","timestamp":1782860163000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00138-026-01834-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,20]]},"references-count":72,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["1834"],"URL":"https:\/\/doi.org\/10.1007\/s00138-026-01834-9","relation":{},"ISSN":["0932-8092","1432-1769"],"issn-type":[{"value":"0932-8092","type":"print"},{"value":"1432-1769","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,20]]},"assertion":[{"value":"19 May 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 February 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 May 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 May 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"74"}}