{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,3]],"date-time":"2026-03-03T16:13:20Z","timestamp":1772554400494,"version":"3.50.1"},"reference-count":70,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2023,6,22]],"date-time":"2023-06-22T00:00:00Z","timestamp":1687392000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,6,22]],"date-time":"2023-06-22T00:00:00Z","timestamp":1687392000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61873086"],"award-info":[{"award-number":["61873086"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Science and Technology Support Program of Changzhou","award":["CE20215022"],"award-info":[{"award-number":["CE20215022"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2024,1]]},"DOI":"10.1007\/s11042-023-15845-5","type":"journal-article","created":{"date-parts":[[2023,6,22]],"date-time":"2023-06-22T10:14:52Z","timestamp":1687428892000},"page":"12159-12184","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":12,"title":["An improved dense-to-sparse cross-modal fusion network for 3D object detection in RGB-D images"],"prefix":"10.1007","volume":"83","author":[{"given":"Yan","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7130-8331","authenticated-orcid":false,"given":"Jianjun","family":"Ni","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangyi","family":"Tang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weidong","family":"Cao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Simon X.","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,6,22]]},"reference":[{"issue":"8","key":"15845_CR1","doi-asserted-by":"publisher","first-page":"373","DOI":"10.1080\/01691864.2022.2043183","volume":"36","author":"R Araki","year":"2022","unstructured":"Araki R, Hirakawa T, Yamashita T, Fujiyoshi H (2022) MT-DSSD: multi-task deconvolutional single shot detector for object detection, segmentation, and grasping detection. Advanced Robotics 36(8):373\u2013387. https:\/\/doi.org\/10.1080\/01691864.2022.2043183","journal-title":"Advanced Robotics"},{"key":"15845_CR2","doi-asserted-by":"publisher","unstructured":"Bai, X, Hu, Z, Zhu, X, Huang, Q, Chen, Y, Fu, H, Tai, C-L (2022) Transfusion: Robust lidar-camera fusion for 3D object detection with transformers. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR):New Orleans, LA, USA, pp 090\u20131099. https:\/\/doi.org\/10.1109\/CVPR52688.2022.00116","DOI":"10.1109\/CVPR52688.2022.00116"},{"key":"15845_CR3","doi-asserted-by":"publisher","unstructured":"Chang, J.-R, Chen, Y-S (2018) Pyramid stereo matching network. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition, Salt Lake City, UT, United States, pp 5410\u20135418. https:\/\/doi.org\/10.1109\/CVPR.2018.00567","DOI":"10.1109\/CVPR.2018.00567"},{"key":"15845_CR4","doi-asserted-by":"publisher","unstructured":"Chen, Z, Huang, S, Tao, D (2018) Context refinement for object detection. In: Lecture Notes in Computer Science (including Subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics):vol 11212 LNCS. Munich, Germany, pp 74\u201389. https:\/\/doi.org\/10.1007\/978-3-030-01237-3_5","DOI":"10.1007\/978-3-030-01237-3_5"},{"key":"15845_CR5","doi-asserted-by":"publisher","unstructured":"Chen, J, Lei, B, Song, Q, Ying, H, Chen, DZ, Wu, J (2020) A hierarchical graph network for 3D object detection on point clouds. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition, Virtual, Online, United States, pp 389\u2013398. https:\/\/doi.org\/10.1109\/CVPR42600.2020.00047","DOI":"10.1109\/CVPR42600.2020.00047"},{"key":"15845_CR6","doi-asserted-by":"crossref","unstructured":"Chen, Z, Li, Z, Zhang, S, Fang, L, Jiang, Q, Zhao, F (2022) AutoAlignV2: Deformable feature aggregation for dynamic multi-modal 3D object detection. arXiv:2207.10316https:\/\/doi.org\/10.48550","DOI":"10.1007\/978-3-031-20074-8_36"},{"key":"15845_CR7","doi-asserted-by":"publisher","unstructured":"Cheng, B, Sheng, L, Shi, S, Yang, M, Xu, D (2021) Back-tracing representative points for voting-based 3D object detection in point clouds. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition, Virtual, Online, United States, pp 8959\u20138968. https:\/\/doi.org\/10.1109\/CVPR46437.2021.00885","DOI":"10.1109\/CVPR46437.2021.00885"},{"key":"15845_CR8","doi-asserted-by":"publisher","unstructured":"Dai, A, Chang, AX, Savva, M, Halber, M, Funkhouser, T, Niecner, M (2017) ScanNet: Richly-annotated 3D reconstructions of indoor scenes. In: Proceedings - 30th IEEE conference on computer vision and pattern recognition, CVPR 2017, vol 2017-January. Honolulu, HI, United States, pp 2432\u20132443. https:\/\/doi.org\/10.1109\/CVPR.2017.261","DOI":"10.1109\/CVPR.2017.261"},{"key":"15845_CR9","doi-asserted-by":"publisher","unstructured":"Ding, M, Huo, Y, Yi, H, Wang, Z, Shi, J, Lu, Z, Luo, P (2020) Learning depth-guided convolutions for monocular 3d object detection. In: Proceedings of the IEEE computer society conference on computer Vision and Pattern Recognition, Virtual, Online, United States, pp 11669\u201311678. https:\/\/doi.org\/10.1109\/CVPR42600.2020.01169","DOI":"10.1109\/CVPR42600.2020.01169"},{"key":"15845_CR10","doi-asserted-by":"publisher","unstructured":"Engelcke, M, Rao, D, Wang, D.Z, Tong, C.H, Posner, I (2017) Vote3Deep: Fast object detection in 3D point clouds using efficient convolutional neural networks. In: Proceedings - IEEE international conference on robotics and automation, vol 0. Singapore, Singapore, pp 1355\u20131361. https:\/\/doi.org\/10.1109\/ICRA.2017.7989161","DOI":"10.1109\/ICRA.2017.7989161"},{"key":"15845_CR11","doi-asserted-by":"publisher","unstructured":"Fu, H, Gong, M, Wang, C, Batmanghelich, K, Tao, D (2018) Deep ordinal regression network for monocular depth estimation. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition, Salt Lake City, UT, United States, pp 2002\u20132011. https:\/\/doi.org\/10.1109\/CVPR.2018.00214","DOI":"10.1109\/CVPR.2018.00214"},{"key":"15845_CR12","doi-asserted-by":"publisher","DOI":"10.1016\/j.displa.2020.101972","volume":"65","author":"Z Gao","year":"2020","unstructured":"Gao Z, Zhai G, Deng H, Yang X (2020) Extended geometric models for stereoscopic 3D with vertical screen disparity. Displays 65:101972. https:\/\/doi.org\/10.1016\/j.displa.2020.101972","journal-title":"Displays"},{"key":"15845_CR13","doi-asserted-by":"publisher","unstructured":"Gupta, S, Arbelaez, P, Girshick, R, Malik, J (2015) Aligning 3D models to RGB-D images of cluttered scenes. In: Proceedings of the IEEE Computer Society Conference on Computer Vision and Pattern Recognition, vol 07-12-June-2015. Boston, MA, United States, pp 4731\u20134740. https:\/\/doi.org\/10.1109\/CVPR.2015.7299105","DOI":"10.1109\/CVPR.2015.7299105"},{"key":"15845_CR14","doi-asserted-by":"publisher","unstructured":"Gupta, S, Girshick, R, Arbelaez, P, Malik, J (2014) Learning rich features from RGB-D images for object detection and segmentation. In: Lecture notes in computer science (including subseries lecture notes in artificial intelligence and lecture notes in bioinformatics):vol 8695 LNCS. Zurich, Switzerland, pp 345\u2013360. https:\/\/doi.org\/10.1007\/978-3-319-10584-0_23","DOI":"10.1007\/978-3-319-10584-0_23"},{"key":"15845_CR15","doi-asserted-by":"publisher","unstructured":"Huang, S, Xie, Y, Zhu, S.-C, Zhu, Y (2021) Spatio-temporal self-supervised representation learning for 3D point clouds. In: Proceedings of the IEEE International Conference on Computer Vision, Virtual, Online, Canada, pp 6515\u20136525. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00647","DOI":"10.1109\/ICCV48922.2021.00647"},{"issue":"45\u201346","key":"15845_CR16","doi-asserted-by":"publisher","first-page":"34129","DOI":"10.1007\/s11042-020-09232-7","volume":"79","author":"G Jeon","year":"2020","unstructured":"Jeon G, Anisetti M, Damiani E, Kantarci B (2020) Artificial intelligence in deep learning algorithms for multimedia analysis. Multimedia Tools and Applications 79(45\u201346):34129\u201334139. https:\/\/doi.org\/10.1007\/s11042-020-09232-7","journal-title":"Multimedia Tools and Applications"},{"issue":"4","key":"15845_CR17","doi-asserted-by":"publisher","first-page":"5973","DOI":"10.1007\/s11042-021-11801-3","volume":"81","author":"C Ji","year":"2022","unstructured":"Ji C, Liu G, Zhao D (2022) Monocular 3D object detection via estimation of paired keypoints for autonomous driving. Multimedia Tools and Applications 81(4):5973\u20135988. https:\/\/doi.org\/10.1007\/s11042-021-11801-3","journal-title":"Multimedia Tools and Applications"},{"key":"15845_CR18","doi-asserted-by":"publisher","unstructured":"Keselman, L, Woodfill, JI, Grunnet-Jepsen, A, Bhowmik, A (2017) Intel(R) RealSense(TM) stereoscopic depth cameras. In: IEEE computer society conference on computer vision and pattern recognition workshops, vol 2017-July. Honolulu, HI, United States, pp 1267\u20131276. https:\/\/doi.org\/10.1109\/CVPRW.2017.167","DOI":"10.1109\/CVPRW.2017.167"},{"key":"15845_CR19","doi-asserted-by":"publisher","unstructured":"Ku, J, Mozifian, M, Lee, J, Harakeh, A, Waslander, SL (2018) Joint 3D proposal generation and object detection from view aggregation. In: IEEE International Conference on Intelligent Robots and Systems, Madrid, Spain, pp 5750\u20135757. https:\/\/doi.org\/10.1109\/IROS.2018.8594049","DOI":"10.1109\/IROS.2018.8594049"},{"key":"15845_CR20","doi-asserted-by":"publisher","unstructured":"Lahoud, J, Ghanem, B (2017) 2D-Driven 3D object detection in RGB-D images. In: Proceedings of the IEEE International Conference on Computer Vision, vol 2017-October. Venice, Italy, pp 4632\u20134640. https:\/\/doi.org\/10.1109\/ICCV.2017.495","DOI":"10.1109\/ICCV.2017.495"},{"key":"15845_CR21","doi-asserted-by":"publisher","unstructured":"Li, B, Ouyang, W, Sheng, L, Zeng, X, Wang, X (2020) GS3D: An efficient 3D object detection framework for autonomous driving. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition, vol 2019-June. Long Beach, CA, United States, pp 1019\u20131028. https:\/\/doi.org\/10.1109\/CVPR.2019.00111","DOI":"10.1109\/CVPR.2019.00111"},{"key":"15845_CR22","doi-asserted-by":"publisher","unstructured":"Li, Y, Qi, X, Chen, Y, Wang, L, Li, Z, Sun, J, Jia, J (2022) Voxel field fusion for 3D object detection. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR):New Orleans, LA, USA, pp 1120\u20131129. https:\/\/doi.org\/10.1109\/CVPR52688.2022.00119","DOI":"10.1109\/CVPR52688.2022.00119"},{"issue":"4","key":"15845_CR23","doi-asserted-by":"publisher","first-page":"985","DOI":"10.1109\/TMM.2017.2759508","volume":"20","author":"J Li","year":"2018","unstructured":"Li J, Liang X, Shen S, Xu T, Feng J, Yan S (2018) Scale-aware Fast R-CNN for pedestrian detection. IEEE Transactions on Multimedia 20(4):985\u2013996. https:\/\/doi.org\/10.1109\/TMM.2017.2759508","journal-title":"IEEE Transactions on Multimedia"},{"key":"15845_CR24","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1016\/j.isprsjprs.2020.05.008","volume":"165","author":"Y Li","year":"2020","unstructured":"Li Y, Ma L, Tan W, Sun C, Cao D, Li J (2020) GRNet: Geometric relation network for 3D object detection from point clouds. ISPRS Journal of Photogrammetry and Remote Sensing 165:43\u201353. https:\/\/doi.org\/10.1016\/j.isprsjprs.2020.05.008","journal-title":"ISPRS Journal of Photogrammetry and Remote Sensing"},{"issue":"1","key":"15845_CR25","doi-asserted-by":"publisher","first-page":"589","DOI":"10.1109\/TKDE.2021.3082470","volume":"35","author":"L Li","year":"2021","unstructured":"Li L, Wan Z, He H (2021) Incomplete multi-view clustering with joint partition and graph learning. IEEE Transactions on Knowledge and Data Engineering 35(1):589\u2013602. https:\/\/doi.org\/10.1109\/TKDE.2021.3082470","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"15845_CR26","doi-asserted-by":"publisher","unstructured":"Liu, Z, Zhang, Z, Cao, Y, Hu, H, Tong, X (2021) Group-free 3D object detection via transformers. In: Proceedings of the IEEE international conference on computer vision, Virtual, Online, Canada, pp 2929\u20132938. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00294","DOI":"10.1109\/ICCV48922.2021.00294"},{"issue":"5","key":"15845_CR27","doi-asserted-by":"publisher","first-page":"707","DOI":"10.1007\/s00371-017-1408-3","volume":"34","author":"B Liu","year":"2018","unstructured":"Liu B, Wu H, Su W, Zhang W, Sun J (2018) Rotation-invariant object detection using sector-ring HOG and boosted random ferns. Visual Computer 34(5):707\u2013719. https:\/\/doi.org\/10.1007\/s00371-017-1408-3","journal-title":"Visual Computer"},{"key":"15845_CR28","doi-asserted-by":"publisher","first-page":"70","DOI":"10.1016\/j.neucom.2022.09.117","volume":"513","author":"Y-F Lu","year":"2022","unstructured":"Lu Y-F, Yu Q, Gao J-W, Li Y, Zou J-C, Qiao H (2022) Cross stage partial connections based weighted bi-directional feature pyramid and enhanced spatial transformation network for robust object detection. Neurocomputing 513:70\u201382. https:\/\/doi.org\/10.1016\/j.neucom.2022.09.117","journal-title":"Neurocomputing"},{"key":"15845_CR29","doi-asserted-by":"publisher","unstructured":"Luo, S, Dai, H, Shao, L, Ding, Y (2021) M3DSSD: Monocular 3D single stage object detector. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition, Virtual, Online, United States, pp 6141\u20136150. https:\/\/doi.org\/10.1109\/CVPR46437.2021.00608","DOI":"10.1109\/CVPR46437.2021.00608"},{"key":"15845_CR30","doi-asserted-by":"publisher","first-page":"364","DOI":"10.1016\/j.neucom.2019.10.025","volume":"378","author":"Q Luo","year":"2020","unstructured":"Luo Q, Ma H, Tang L, Wang Y, Xiong R (2020) 3D-SSD: Learning hierarchical features from RGB-D images for amodal 3D object detection. Neurocomputing 378:364\u2013374. https:\/\/doi.org\/10.1016\/j.neucom.2019.10.025","journal-title":"Neurocomputing"},{"key":"15845_CR31","doi-asserted-by":"publisher","unstructured":"Misra, I, Girdhar, R, Joulin, A (2021) An end-to-end transformer model for 3D object detection. In: Proceedings of the IEEE international conference on computer vision, Virtual, Online, Canada, pp 2886\u20132897. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00290","DOI":"10.1109\/ICCV48922.2021.00290"},{"key":"15845_CR32","doi-asserted-by":"publisher","unstructured":"Mousavian, A, Anguelov, D, Koecka, J, Flynn, J (2017) 3D bounding box estimation using deep learning and geometry. In: Proceedings - 30th IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017, vol. 2017-January. Honolulu, HI, United States, pp 5632\u20135640. https:\/\/doi.org\/10.1109\/CVPR.2017.597","DOI":"10.1109\/CVPR.2017.597"},{"issue":"8","key":"15845_CR33","doi-asserted-by":"publisher","first-page":"2749","DOI":"10.3390\/app10082749","volume":"10","author":"J Ni","year":"2020","unstructured":"Ni J, Chen Y, Chen Y, Zhu J, Ali D, Cao W (2020) A survey on theories and applications for self-driving cars based on deep learning methods. Applied Sciences-Basel 10(8):2749. https:\/\/doi.org\/10.3390\/app10082749","journal-title":"Applied Sciences-Basel"},{"key":"15845_CR34","doi-asserted-by":"publisher","first-page":"5001614","DOI":"10.1109\/TIM.2022.3146923","volume":"71","author":"J Ni","year":"2022","unstructured":"Ni J, Shen K, Chen Y, Cao W, Yang SX (2022) An improved deep network-based scene classification method for self-driving cars. IEEE Transactions on Instrumentation and Measurement 71:5001614. https:\/\/doi.org\/10.1109\/TIM.2022.3146923","journal-title":"IEEE Transactions on Instrumentation and Measurement"},{"key":"15845_CR35","doi-asserted-by":"publisher","unstructured":"Qi, C.R, Chen, X, Litany, O, Guibas, LJ (2020) ImVoteNet: Boosting 3D object detection in point clouds with image votes. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition, Virtual, Online, United States, pp 4403\u20134412. https:\/\/doi.org\/10.1109\/CVPR42600.2020.00446","DOI":"10.1109\/CVPR42600.2020.00446"},{"key":"15845_CR36","doi-asserted-by":"publisher","unstructured":"Qi, C.R, Litany, O, He, K, Guibas, L (2019) Deep hough voting for 3D object detection in point clouds. In: Proceedings of the IEEE international conference on computer vision, vol 2019-October. Seoul, Korea, Republic of, pp 9276\u20139285. https:\/\/doi.org\/10.1109\/ICCV.2019.00937","DOI":"10.1109\/ICCV.2019.00937"},{"key":"15845_CR37","doi-asserted-by":"publisher","unstructured":"Qi, C.R, Liu, W, Wu, C, Su, H, Guibas, LJ (2018) Frustum pointnets for 3D object detection from RGB-D data. In: Proceedings of the IEEE Computer Society Conference on Computer Vision and Pattern Recognition, Salt Lake City, UT, United States, pp 918\u2013927. https:\/\/doi.org\/10.1109\/CVPR.2018.00102","DOI":"10.1109\/CVPR.2018.00102"},{"key":"15845_CR38","doi-asserted-by":"publisher","unstructured":"Qi, C.R, Su, H, Mo, K, Guibas, LJ (2017) PointNet: Deep learning on point sets for 3D classification and segmentation. In: Proceedings - 30th IEEE conference on computer vision and pattern recognition, CVPR 2017, vol 2017-January. Honolulu, HI, United States, pp 77\u201385. https:\/\/doi.org\/10.1109\/CVPR.2017.16","DOI":"10.1109\/CVPR.2017.16"},{"key":"15845_CR39","first-page":"5100","volume-title":"Advances in neural information processing systems, vol 2017-December","author":"CR Qi","year":"2017","unstructured":"Qi CR, Yi L, Su H, Guibas LJ (2017) PointNet++: Deep hierarchical feature learning on point sets in a metric space. Advances in neural information processing systems, vol 2017-December. Long Beach, CA, United States, pp 5100\u20135109"},{"key":"15845_CR40","doi-asserted-by":"publisher","first-page":"2947","DOI":"10.1109\/TIP.2019.2955239","volume":"29","author":"MM Rahman","year":"2020","unstructured":"Rahman MM, Tan Y, Xue J, Lu K (2020) Notice of removal: Recent advances in 3d object detection in the era of deep neural networks: A survey. IEEE Transactions on Image Processing 29:2947\u20132962. https:\/\/doi.org\/10.1109\/TIP.2019.2955239","journal-title":"IEEE Transactions on Image Processing"},{"issue":"10","key":"15845_CR41","doi-asserted-by":"publisher","first-page":"2670","DOI":"10.1109\/TPAMI.2019.2923201","volume":"42","author":"Z Ren","year":"2020","unstructured":"Ren Z, Sudderth EB (2020) Clouds of oriented gradients for 3D detection of objects, surfaces, and indoor scene layouts. IEEE Transactions on Pattern Analysis and Machine Intelligence 42(10):2670\u20132683. https:\/\/doi.org\/10.1109\/TPAMI.2019.2923201","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"15845_CR42","doi-asserted-by":"publisher","first-page":"131","DOI":"10.1016\/j.jvcir.2018.05.019","volume":"55","author":"Y Ren","year":"2018","unstructured":"Ren Y, Chen C, Li S, Kuo C-CJ (2018) Context-assisted 3D (C3D) object detection from RGB-D images. Journal of Visual Communication and Image Representation 55:131\u2013141. https:\/\/doi.org\/10.1016\/j.jvcir.2018.05.019","journal-title":"Journal of Visual Communication and Image Representation"},{"issue":"1","key":"15845_CR43","doi-asserted-by":"publisher","first-page":"105","DOI":"10.1109\/TPAMI.2008.275","volume":"32","author":"E Rosten","year":"2010","unstructured":"Rosten E, Porter R, Drummond T (2010) Faster and better: A machine learning approach to corner detection. IEEE Transactions on Pattern Analysis and Machine Intelligence 32(1):105\u2013119. https:\/\/doi.org\/10.1109\/TPAMI.2008.275","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"15845_CR44","doi-asserted-by":"publisher","unstructured":"Shi, S, Wang, X, Li, H (2019) PointRCNN: 3D object proposal generation and detection from point cloud. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition, vol 2019-June. Long Beach, CA, United States, pp 770\u2013779. https:\/\/doi.org\/10.1109\/CVPR.2019.00086","DOI":"10.1109\/CVPR.2019.00086"},{"key":"15845_CR45","doi-asserted-by":"publisher","unstructured":"Silberman, N, Hoiem, D, Kohli, P, Fergus, R (2012) Indoor segmentation and support inference from RGBD images. In: Lecture notes in computer science (including Subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics):vol 7576 LNCS. Florence, Italy, pp 746\u2013760. https:\/\/doi.org\/10.1007\/978-3-642-33715-4_54","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"15845_CR46","doi-asserted-by":"publisher","unstructured":"Song, S, Lichtenberg, S.P, Xiao, J (2015) SUN RGB-D: A RGB-D scene understanding benchmark suite. In: Proceedings of the IEEE Computer Society Conference on Computer Vision and Pattern Recognition, vol 07-12-June-2015. Boston, MA, United States, pp 567\u2013576. https:\/\/doi.org\/10.1109\/CVPR.2015.7298655","DOI":"10.1109\/CVPR.2015.7298655"},{"key":"15845_CR47","doi-asserted-by":"publisher","unstructured":"Song, S, Xiao, J (2014) Sliding shapes for 3D object detection in depth images. In: Lecture notes in computer science (including subseries lecture notes in artificial intelligence and lecture notes in bioinformatics):vol 8694 LNCS. Zurich, Switzerland, pp 634\u2013651. https:\/\/doi.org\/10.1007\/978-3-319-10599-4_41","DOI":"10.1007\/978-3-319-10599-4_41"},{"key":"15845_CR48","doi-asserted-by":"publisher","unstructured":"Song, S, Xiao, J (2016) Deep sliding shapes for amodal 3D object detection in RGB-D images. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition, vol 2016-December. Las Vegas, NV, United States, pp 808\u2013816. https:\/\/doi.org\/10.1109\/CVPR.2016.94","DOI":"10.1109\/CVPR.2016.94"},{"key":"15845_CR49","doi-asserted-by":"publisher","unstructured":"Sun, R, Qian, J, Jose, R.H, Gong, Z, Miao, R, Xue, W, Liu, P (2020) A flexible and efficient real-time ORB-based full-HD image feature extraction accelerator. IEEE Transactions on Very Large Scale Integration (VLSI) Systems, 28(2):565\u2013575. https:\/\/doi.org\/10.1109\/TVLSI.2019.2945982","DOI":"10.1109\/TVLSI.2019.2945982"},{"key":"15845_CR50","first-page":"5999","volume-title":"Advances in neural information processing systems, vol 2017-December","author":"A Vaswani","year":"2017","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser L, Polosukhin I (2017) Attention is all you need. Advances in neural information processing systems, vol 2017-December. Long Beach, CA, United States, pp 5999\u20136009"},{"key":"15845_CR51","doi-asserted-by":"publisher","unstructured":"Wang, Y, Chen, X, Cao, L, Huang, W, Sun, F, Wang, Y (2022) Multimodal token fusion for vision transformers. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR):New Orleans, LA, USA, pp 12186\u201312195. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01187","DOI":"10.1109\/CVPR52688.2022.01187"},{"key":"15845_CR52","doi-asserted-by":"publisher","unstructured":"Wang, H, Shi, S, Yang, Z, Fang, R, Qian, Q, Li, H, Schiele, B, Wang, L (2022) RBGNet: Ray-based grouping for 3D object detection. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR):New Orleans, LA, USA, pp 1110\u20131119. https:\/\/doi.org\/10.1109\/CVPR52688.2022.00118","DOI":"10.1109\/CVPR52688.2022.00118"},{"key":"15845_CR53","doi-asserted-by":"publisher","unstructured":"Wang, W, Tran, D, Feiszli, M (2020) What makes training multi-modal classification networks hard? In: Proceedings of the IEEE Computer Society Conference on Computer Vision and Pattern Recognition, Virtual, Online, United States, pp 12692\u201312702. https:\/\/doi.org\/10.1109\/CVPR42600.2020.01271","DOI":"10.1109\/CVPR42600.2020.01271"},{"key":"15845_CR54","doi-asserted-by":"publisher","unstructured":"Wang, Y, Ye, T, Cao, L, Huang, W, Sun, F, He, F, Tao, D (2022) Bridged transformer for vision and point cloud 3D object detection. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR):New Orleans, LA, USA, pp 12114\u201312123. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01180","DOI":"10.1109\/CVPR52688.2022.01180"},{"key":"15845_CR55","doi-asserted-by":"publisher","DOI":"10.1016\/j.displa.2021.102077","volume":"70","author":"Y Wang","year":"2021","unstructured":"Wang Y, Wang C, Long P, Gu Y, Li W (2021) Recent advances in 3D object detection based on RGB-D: A survey. Displays 70:102077. https:\/\/doi.org\/10.1016\/j.displa.2021.102077","journal-title":"Displays"},{"issue":"1","key":"15845_CR56","doi-asserted-by":"publisher","first-page":"6","DOI":"10.1145\/3462219","volume":"18","author":"Z Wang","year":"2022","unstructured":"Wang Z, Xie Q, Wei M, Long K, Wang J (2022) Multi-feature fusion VoteNet for 3D object detection. ACM Transactions on Multimedia Computing, Communications and Applications 18(1):6. https:\/\/doi.org\/10.1145\/3462219","journal-title":"ACM Transactions on Multimedia Computing, Communications and Applications"},{"issue":"3","key":"15845_CR57","doi-asserted-by":"publisher","first-page":"332","DOI":"10.1007\/s11263-013-0623-2","volume":"106","author":"OJ Woodford","year":"2014","unstructured":"Woodford OJ, Pham M-T, Maki A, Perbet F, Stenger B (2014) Demisting the hough transform for 3d shape recognition and registration. International Journal of Computer Vision 106(3):332\u2013341. https:\/\/doi.org\/10.1007\/s11263-013-0623-2","journal-title":"International Journal of Computer Vision"},{"key":"15845_CR58","doi-asserted-by":"publisher","unstructured":"Xiao, J, Owens, A, Torralba, A (2013) SUN3D: A database of big spaces reconstructed using SfM and object labels. In: Proceedings of the IEEE international conference on computer vision, Sydney, NSW, Australia, pp 1625\u20131632. https:\/\/doi.org\/10.1109\/ICCV.2013.458","DOI":"10.1109\/ICCV.2013.458"},{"issue":"33\u201334","key":"15845_CR59","doi-asserted-by":"publisher","first-page":"23729","DOI":"10.1007\/s11042-020-08976-6","volume":"79","author":"Y Xiao","year":"2020","unstructured":"Xiao Y, Tian Z, Yu J, Zhang Y, Liu S, Du S, Lan X (2020) A review of object detection based on deep learning. Multimedia Tools and Applications 79(33\u201334):23729\u201323791. https:\/\/doi.org\/10.1007\/s11042-020-08976-6","journal-title":"Multimedia Tools and Applications"},{"issue":"6","key":"15845_CR60","doi-asserted-by":"publisher","first-page":"1857","DOI":"10.1007\/s11263-021-01456-w","volume":"129","author":"Q Xie","year":"2021","unstructured":"Xie Q, Lai Y-K, Wu J, Wang Z, Zhang Y, Xu K, Wang J (2021) Vote-based 3D object detection with context modeling and SOB-3DNMS. International Journal of Computer Vision 129(6):1857\u20131874. https:\/\/doi.org\/10.1007\/s11263-021-01456-w","journal-title":"International Journal of Computer Vision"},{"key":"15845_CR61","doi-asserted-by":"publisher","unstructured":"Xu, D, Anguelov, D, Jain, A (2018) PointFusion: Deep sensor fusion for 3D bounding box estimation. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition, Salt Lake City, UT, United States, pp 244\u2013253. https:\/\/doi.org\/10.1109\/CVPR.2018.00033","DOI":"10.1109\/CVPR.2018.00033"},{"key":"15845_CR62","doi-asserted-by":"publisher","unstructured":"Xu, B, Chen, Z (2018) Multi-level fusion based 3D object detection from monocular images. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition, Salt Lake City, UT, United States, pp 2345\u20132353. https:\/\/doi.org\/10.1109\/CVPR.2018.00249","DOI":"10.1109\/CVPR.2018.00249"},{"key":"15845_CR63","doi-asserted-by":"publisher","unstructured":"Zhang, Y, Chen, J, Huang, D (2022) CAT-Det: Contrastively augmented transformer for multi-modal 3D object detection. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR):New Orleans, LA, USA, pp 908\u2013917. https:\/\/doi.org\/10.1109\/CVPR52688.2022.00098","DOI":"10.1109\/CVPR52688.2022.00098"},{"issue":"2","key":"15845_CR64","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1109\/MMUL.2012.24","volume":"19","author":"Z Zhang","year":"2012","unstructured":"Zhang Z (2012) Microsoft kinect sensor and its effect. IEEE Multimedia 19(2):4\u201310. https:\/\/doi.org\/10.1109\/MMUL.2012.24","journal-title":"IEEE Multimedia"},{"issue":"22","key":"15845_CR65","doi-asserted-by":"publisher","first-page":"4706","DOI":"10.3390\/rs13224706","volume":"13","author":"M Zhang","year":"2021","unstructured":"Zhang M, Xu S, Song W, He Q (2021) Wei, Q (2021) Lightweight underwater object detection based on YOLO v4 and multi-scale attentional feature fusion. Remote Sensing 13(22):4706. https:\/\/doi.org\/10.3390\/rs13224706","journal-title":"Remote Sensing"},{"key":"15845_CR66","doi-asserted-by":"publisher","DOI":"10.1016\/j.displa.2021.102022","volume":"68","author":"L Zhang","year":"2021","unstructured":"Zhang L, Li W, Yu L, Sun L, Dong X, Ning X (2021) GmFace: An explicit function for face image representation. Displays 68:102022. https:\/\/doi.org\/10.1016\/j.displa.2021.102022","journal-title":"Displays"},{"issue":"12","key":"15845_CR67","doi-asserted-by":"publisher","first-page":"4735","DOI":"10.1109\/TCSVT.2021.3102025","volume":"31","author":"L Zhao","year":"2021","unstructured":"Zhao L, Guo J, Xu D, Sheng L (2021) Transformer3D-Det: Improving 3D object detection by vote refinement. IEEE Transactions on Circuits and Systems for Video Technology 31(12):4735\u20134746. https:\/\/doi.org\/10.1109\/TCSVT.2021.3102025","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"15845_CR68","doi-asserted-by":"publisher","unstructured":"Zhou, Z, Fan, X, Shi, P, Xin, Y (2021) R-MSFM: Recurrent multi-scale feature modulation for monocular depth estimating. In: Proceedings of the IEEE international conference on computer vision, Virtual, Online, Canada, pp 12757\u201312766. https:\/\/doi.org\/10.1109\/ICCV48922.2021.01254","DOI":"10.1109\/ICCV48922.2021.01254"},{"key":"15845_CR69","doi-asserted-by":"publisher","unstructured":"Zhou, Y, Tuzel, O (2018) VoxelNet: End-to-end learning for point cloud based 3D object detection. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition, Salt Lake City, UT, United States, pp 4490\u20134499. https:\/\/doi.org\/10.1109\/CVPR.2018.00472","DOI":"10.1109\/CVPR.2018.00472"},{"issue":"3","key":"15845_CR70","doi-asserted-by":"publisher","first-page":"345","DOI":"10.1016\/j.cviu.2008.08.006","volume":"113","author":"H Zhou","year":"2009","unstructured":"Zhou H, Yuan Y, Shi C (2009) Object tracking using SIFT features and mean shift. Computer Vision and Image Understanding 113(3):345\u2013352. https:\/\/doi.org\/10.1016\/j.cviu.2008.08.006","journal-title":"Computer Vision and Image Understanding"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-15845-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-15845-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-15845-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,10]],"date-time":"2024-01-10T09:21:29Z","timestamp":1704878489000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-15845-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,22]]},"references-count":70,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2024,1]]}},"alternative-id":["15845"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-15845-5","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,6,22]]},"assertion":[{"value":"22 December 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 April 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 May 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 June 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declared that they have no conflicts of interest to this work.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interest"}}]}}