{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T16:13:02Z","timestamp":1784563982051,"version":"3.55.0"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2025,6,2]],"date-time":"2025-06-02T00:00:00Z","timestamp":1748822400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,6,2]],"date-time":"2025-06-02T00:00:00Z","timestamp":1748822400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["52105031"],"award-info":[{"award-number":["52105031"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s11760-025-04221-5","type":"journal-article","created":{"date-parts":[[2025,6,2]],"date-time":"2025-06-02T12:20:45Z","timestamp":1748866845000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Category-Level 6D Pose Estimation Based on Deep Cross-Modal Feature Fusion"],"prefix":"10.1007","volume":"19","author":[{"given":"Chunhui","family":"Tang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingyang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yi","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shouxue","family":"Shan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,6,2]]},"reference":[{"key":"4221_CR1","doi-asserted-by":"crossref","unstructured":"Mousavian, A., Eppner, C., Fox, D.: 6-DOF GraspNet: Variational Grasp Generation for Object Manipulation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2901\u20132910 (2019)","DOI":"10.1109\/ICCV.2019.00299"},{"key":"4221_CR2","doi-asserted-by":"publisher","first-page":"276","DOI":"10.1016\/j.rcim.2018.10.001","volume":"56","author":"M Gattullo","year":"2019","unstructured":"Gattullo, M., Scurati, G.W., Fiorentino, M., Uva, A.E., Ferrise, F., Bordegoni, M.: Towards augmented reality manuals for industry 4.0: A methodology. Robotics and computer-integrated manufacturing 56, 276\u2013286 (2019)","journal-title":"Robotics and computer-integrated manufacturing"},{"key":"4221_CR3","doi-asserted-by":"publisher","first-page":"2086","DOI":"10.3389\/fpsyg.2018.02086","volume":"9","author":"P Cipresso","year":"2018","unstructured":"Cipresso, P., Giglioli, I.A.C., Raya, M.A., Riva, G.: The Past, Present, and Future of Virtual and Augmented Reality Research: A Network and Cluster Analysis of the Literature. Frontiers in psychology 9, 2086 (2018)","journal-title":"Frontiers in psychology"},{"key":"4221_CR4","doi-asserted-by":"crossref","unstructured":"He, Y., Sun, W., Huang, H., Liu, J., Fan, H., Sun, J.: PVN3D: A Deep Point-wise 3D Keypoints Voting Network for 6DoF Pose Estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11632\u201311641 (2020)","DOI":"10.1109\/CVPR42600.2020.01165"},{"key":"4221_CR5","doi-asserted-by":"crossref","unstructured":"He, Y., Huang, H., Fan, H., Chen, Q., Sun, J.: FFB6D: A Full Flow Bidirectional Fusion Network for 6D Pose Estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3003\u20133013 (2021)","DOI":"10.1109\/CVPR46437.2021.00302"},{"key":"4221_CR6","doi-asserted-by":"crossref","unstructured":"Zhou, J., Chen, K., Xu, L., Dou, Q., Qin, J.: Deep Fusion Transformer Network with Weighted Vector-Wise Keypoints Voting for Robust 6D Object Pose Estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13967\u201313977 (2023)","DOI":"10.1109\/ICCV51070.2023.01284"},{"issue":"8\u20139","key":"4221_CR7","doi-asserted-by":"publisher","first-page":"6309","DOI":"10.1007\/s11760-024-03318-7","volume":"18","author":"Y Song","year":"2024","unstructured":"Song, Y., Tang, C.: A RGB-D Feature Fusion Network for Occluded Object 6D Pose Estimation. Signal, Image and Video Processing 18(8\u20139), 6309\u20136319 (2024)","journal-title":"Signal, Image and Video Processing"},{"key":"4221_CR8","doi-asserted-by":"crossref","unstructured":"Wang, H., Sridhar, S., Huang, J., Valentin, J., Song, S., Guibas, L.J.: Normalized Object Coordinate Space for Category-Level 6D Object Pose and Size Estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2642\u20132651 (2019)","DOI":"10.1109\/CVPR.2019.00275"},{"key":"4221_CR9","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask R-CNN. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2961\u20132969 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"4221_CR10","doi-asserted-by":"crossref","unstructured":"Chen, K., Dou, Q.: SGPA: Structure-Guided Prior Adaptation for Category-Level 6D Object Pose Estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2773\u20132782 (2021)","DOI":"10.1109\/ICCV48922.2021.00277"},{"key":"4221_CR11","doi-asserted-by":"crossref","unstructured":"Liu, J., Chen, Y., Ye, X., Qi, X.: Prior-free Category-level Pose Estimation with Implicit Space Transformation. In: IEEE International Conference on Computer Vision 2023 (02\/10\/2023-06\/10\/2023, Paris) (2023)","DOI":"10.1109\/ICCV51070.2023.01285"},{"key":"4221_CR12","doi-asserted-by":"crossref","unstructured":"Lin, H., Liu, Z., Cheang, C., Fu, Y., Guo, G., Xue, X.: Sar-Net: Shape Alignment and Recovery Network for Category-level 6D Object Pose and Size Estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6707\u20136717 (2022)","DOI":"10.1109\/CVPR52688.2022.00659"},{"key":"4221_CR13","doi-asserted-by":"publisher","unstructured":"Umeyama, S.: Least-squares estimation of transformation parameters between two point patterns. IEEE Transactions on Pattern Analysis and Machine Intelligence, 376\u2013380 (1991) https:\/\/doi.org\/10.1109\/34.88573","DOI":"10.1109\/34.88573"},{"key":"4221_CR14","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.-Y., et al.: Segment Anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4015\u20134026 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"4221_CR15","doi-asserted-by":"crossref","unstructured":"Liu, M., Yin, H.: Cross Attention Network for Semantic Segmentation. In: 2019 IEEE International Conference on Image Processing (ICIP), pp. 2434\u20132438 (2019). IEEE","DOI":"10.1109\/ICIP.2019.8803320"},{"key":"4221_CR16","volume-title":"Attention is All you Need","author":"A Vaswani","year":"2017","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A., Kaiser, L., Polosukhin, I.: Attention is All you Need. Neural Information Processing Systems, Neural Information Processing Systems (2017)"},{"key":"4221_CR17","doi-asserted-by":"crossref","unstructured":"Hu, Y., Speierer, S., Jakob, W., Fua, P., Salzmann, M.: Wide-depth-range 6d object pose estimation in space. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15870\u201315879 (2021)","DOI":"10.1109\/CVPR46437.2021.01561"},{"key":"4221_CR18","doi-asserted-by":"crossref","unstructured":"Song, C., Song, J., Huang, Q.: Hybridpose: 6d object pose estimation under hybrid representations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 431\u2013440 (2020)","DOI":"10.1109\/CVPR42600.2020.00051"},{"key":"4221_CR19","doi-asserted-by":"crossref","unstructured":"Wang, C., Xu, D., Zhu, Y., Mart\u00edn-Mart\u00edn, R., Lu, C., Fei-Fei, L., Savarese, S.: DenseFusion: 6D Object Pose Estimation by Iterative Dense Fusion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3343\u20133352 (2019)","DOI":"10.1109\/CVPR.2019.00346"},{"key":"4221_CR20","doi-asserted-by":"crossref","unstructured":"Li, Z., Wang, G., Ji, X.: CDPN: Coordinates-Based Disentangled Pose Network for Real-Time RGB-Based 6-DoF Object Pose Estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7678\u20137687 (2019)","DOI":"10.1109\/ICCV.2019.00777"},{"key":"4221_CR21","doi-asserted-by":"crossref","unstructured":"Sundermeyer, M., Marton, Z.-C., Durner, M., Brucker, M., Triebel, R.: Implicit 3D Orientation Learning for 6D Object Detection from RGB Images. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 699\u2013715 (2018)","DOI":"10.1007\/978-3-030-01231-1_43"},{"key":"4221_CR22","doi-asserted-by":"crossref","unstructured":"Li, H., Lin, J., Jia, K.: DCL-Net: Deep Correspondence Learning Network for 6D Pose Estimation. In: European Conference on Computer Vision, pp. 369\u2013385 (2022). Springer","DOI":"10.1007\/978-3-031-20077-9_22"},{"key":"4221_CR23","doi-asserted-by":"crossref","unstructured":"Tian, M., Pan, L., Ang, M.H., Lee, G.H.: Robust 6D Object Pose Estimation by Learning RGB-D Features. In: 2020 IEEE International Conference on Robotics and Automation (ICRA), pp. 6218\u20136224 (2020). IEEE","DOI":"10.1109\/ICRA40945.2020.9197555"},{"key":"4221_CR24","doi-asserted-by":"crossref","unstructured":"Liu, X., Iwase, S., Kitani, K.M.: KDFNet: Learning Keypoint Distance Field for 6D Object Pose Estimation. In: 2021 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 4631\u20134638 (2021). IEEE","DOI":"10.1109\/IROS51168.2021.9636489"},{"key":"4221_CR25","doi-asserted-by":"crossref","unstructured":"Gao, G., Lauri, M., Wang, Y., Hu, X., Zhang, J., Frintrop, S.: 6D Object Pose Regression via Supervised Learning on Point Clouds. In: 2020 IEEE International Conference on Robotics and Automation (ICRA), pp. 3643\u20133649 (2020). IEEE","DOI":"10.1109\/ICRA40945.2020.9197461"},{"key":"4221_CR26","doi-asserted-by":"crossref","unstructured":"Kleeberger, K., Huber, M.F.: Single Shot 6D Object Pose Estimation. In: 2020 IEEE International Conference on Robotics and Automation (ICRA), pp. 6239\u20136245 (2020). IEEE","DOI":"10.1109\/ICRA40945.2020.9197207"},{"key":"4221_CR27","doi-asserted-by":"crossref","unstructured":"Tian, M., Ang, M.H., Lee, G.H.: Shape Prior Deformation for Categorical 6D Object Pose and Size Estimation. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXI 16, pp. 530\u2013546 (2020). Springer","DOI":"10.1007\/978-3-030-58589-1_32"},{"key":"4221_CR28","unstructured":"Li, G., Li, Y., Ye, Z., Zhang, Q., Kong, T., Cui, Z., Zhang, G.: Generative Category-Level Shape and Pose Estimation with Semantic Primitives. In: Conference on Robot Learning, pp. 1390\u20131400 (2023). PMLR"},{"key":"4221_CR29","doi-asserted-by":"publisher","unstructured":"Lin, Z.-H., Huang, S.-Y., Wang, Y.-C.F.: Convolution in the Cloud: Learning Deformable Kernels in 3D Graph Convolution Networks for Point Cloud Analysis. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020). https:\/\/doi.org\/10.1109\/cvpr42600.2020.00187","DOI":"10.1109\/cvpr42600.2020.00187"},{"issue":"7","key":"4221_CR30","first-page":"579","volume":"8","author":"M-C Popescu","year":"2009","unstructured":"Popescu, M.-C., Balas, V.E., Perescu-Popescu, L., Mastorakis, N.: Multilayer perceptron and neural networks. WSEAS Transactions on Circuits and Systems 8(7), 579\u2013588 (2009)","journal-title":"WSEAS Transactions on Circuits and Systems"},{"key":"4221_CR31","doi-asserted-by":"crossref","unstructured":"Hao, Z., Averbuch-Elor, H., Snavely, N., Belongie, S.: Dualsdf: Semantic Shape Manipulation using a Two-Level Representation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7631\u20137641 (2020)","DOI":"10.1109\/CVPR42600.2020.00765"},{"key":"4221_CR32","doi-asserted-by":"crossref","unstructured":"Cheng, T., Song, L., Ge, Y., Liu, W., Wang, X., Shan, Y.: YOLO-World: Real-Time Open-Vocabulary Object Detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16901\u201316911 (2024)","DOI":"10.1109\/CVPR52733.2024.01599"},{"key":"4221_CR33","first-page":"7537","volume":"33","author":"M Tancik","year":"2020","unstructured":"Tancik, M., Srinivasan, P., Mildenhall, B., Fridovich-Keil, S., Raghavan, N., Singhal, U., Ramamoorthi, R., Barron, J., Ng, R.: Fourier Features Let Networks Learn High Frequency Functions in Low Dimensional Domains. Advances in neural information processing systems 33, 7537\u20137547 (2020)","journal-title":"Advances in neural information processing systems"},{"key":"4221_CR34","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., Uszkoreit, J., Houlsby, N.: An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. ICLR (2021)"},{"key":"4221_CR35","doi-asserted-by":"publisher","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Dollar, P., Girshick, R.: Masked Autoencoders Are Scalable Vision Learners. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2022). https:\/\/doi.org\/10.1109\/cvpr52688.2022.01553","DOI":"10.1109\/cvpr52688.2022.01553"},{"key":"4221_CR36","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep Residual Learning for Image Recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"4221_CR37","doi-asserted-by":"publisher","unstructured":"Zhao, H., Shi, J., Qi, X., Wang, X., Jia, J.: Pyramid Scene Parsing Network. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017). https:\/\/doi.org\/10.1109\/cvpr.2017.660","DOI":"10.1109\/cvpr.2017.660"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-025-04221-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-025-04221-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-025-04221-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,11]],"date-time":"2025-06-11T03:15:39Z","timestamp":1749611739000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-025-04221-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,2]]},"references-count":37,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["4221"],"URL":"https:\/\/doi.org\/10.1007\/s11760-025-04221-5","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-5694014\/v1","asserted-by":"object"}]},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"value":"1863-1703","type":"print"},{"value":"1863-1711","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,6,2]]},"assertion":[{"value":"22 December 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 March 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 April 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 June 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"683"}}