{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T19:06:07Z","timestamp":1757617567432,"version":"3.44.0"},"reference-count":71,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2025,3,11]],"date-time":"2025-03-11T00:00:00Z","timestamp":1741651200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,11]],"date-time":"2025-03-11T00:00:00Z","timestamp":1741651200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No. 62073245"],"award-info":[{"award-number":["No. 62073245"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Shanghai Science and Technology Innovation Action Plan","award":["22511104900"],"award-info":[{"award-number":["22511104900"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s00371-025-03852-6","type":"journal-article","created":{"date-parts":[[2025,3,11]],"date-time":"2025-03-11T00:51:46Z","timestamp":1741654306000},"page":"8009-8023","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["GRPoseNet: a generalizable and robust 6D object pose estimation network using sparse RGB views"],"prefix":"10.1007","volume":"41","author":[{"given":"Wubin","family":"Shi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shaoyan","family":"Gai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feipeng","family":"Da","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zeyu","family":"Cai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiaoling","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,11]]},"reference":[{"key":"3852_CR1","doi-asserted-by":"publisher","first-page":"1143947","DOI":"10.3389\/fpubh.2023.1143947","volume":"11","author":"SG Ali","year":"2023","unstructured":"Ali, S.G., Wang, X., Li, P., Jung, Y., Bi, L., Kim, J., Chen, Y., Feng, D.D., Magnenat Thalmann, N., Wang, J., et al.: A systematic review: virtual-reality-based techniques for human exercises and health improvement. Front. Public Health 11, 1143947 (2023)","journal-title":"Front. Public Health"},{"key":"3852_CR2","doi-asserted-by":"publisher","first-page":"197","DOI":"10.5194\/isprs-annals-V-4-2022-197-2022","volume":"4","author":"A-M Boutsi","year":"2022","unstructured":"Boutsi, A.-M., Bakalos, N., Ioannidis, C.: Pose estimation through mask-r cnn and vslam in large-scale outdoors augmented reality. ISPRS Ann. Photogramm. Remote Sens. Spatial Inf. Sci. 4, 197\u2013204 (2022)","journal-title":"ISPRS Ann. Photogramm. Remote Sens. Spatial Inf. Sci."},{"key":"3852_CR3","doi-asserted-by":"crossref","unstructured":"Chen, H., Wang, P., Wang, F., Tian, W., Xiong, L., Li, H.: Epro-pnp: generalized end-to-end probabilistic perspective-n-points for monocular object pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2781\u20132790 (2022)","DOI":"10.1109\/CVPR52688.2022.00280"},{"key":"3852_CR4","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.119838","volume":"223","author":"S Hoque","year":"2023","unstructured":"Hoque, S., Xu, S., Maiti, A., Wei, Y., Arafat, M.Y.: Deep learning for 6d pose estimation of objects-a case study for autonomous driving. Expert Syst. Appl. 223, 119838 (2023)","journal-title":"Expert Syst. Appl."},{"key":"3852_CR5","doi-asserted-by":"crossref","unstructured":"Wang, C., Mart\u00edn-Mart\u00edn, R., Xu, D., Lv, J., Lu, C., Fei-Fei, L., Savarese, S., Zhu, Y.: 6-pack: category-level 6d pose tracker with anchor-based keypoints. In: 2020 IEEE International Conference on Robotics and Automation (ICRA), pp. 10059\u201310066. IEEE (2020)","DOI":"10.1109\/ICRA40945.2020.9196679"},{"issue":"1","key":"3852_CR6","doi-asserted-by":"publisher","first-page":"137","DOI":"10.1109\/LRA.2022.3222998","volume":"8","author":"M-H Jeon","year":"2022","unstructured":"Jeon, M.-H., Kim, J., Ryu, J.-H., Kim, A.: Ambiguity-aware multi-object pose optimization for visually-assisted robot manipulation. IEEE Robot. Autom. Lett. 8(1), 137\u2013144 (2022)","journal-title":"IEEE Robot. Autom. Lett."},{"key":"3852_CR7","doi-asserted-by":"crossref","unstructured":"Ma, W., Wang, A., Yuille, A., Kortylewski, A.: Robust category-level 6d pose estimation with coarse-to-fine rendering of neural features. In: European Conference on Computer Vision, pp. 492\u2013508. Springer (2022)","DOI":"10.1007\/978-3-031-20077-9_29"},{"key":"3852_CR8","doi-asserted-by":"publisher","first-page":"6907","DOI":"10.1109\/TIP.2022.3216980","volume":"31","author":"L Zou","year":"2022","unstructured":"Zou, L., Huang, Z., Gu, N., Wang, G.: 6d-vit: category-level 6d object pose estimation via transformer-based instance representation learning. IEEE Trans. Image Process. 31, 6907\u20136921 (2022)","journal-title":"IEEE Trans. Image Process."},{"key":"3852_CR9","doi-asserted-by":"crossref","unstructured":"Wang, H., Sridhar, S., Huang, J., Valentin, J., Song, S., Guibas, L.J.: Normalized object coordinate space for category-level 6d object pose and size estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2642\u20132651 (2019)","DOI":"10.1109\/CVPR.2019.00275"},{"issue":"04","key":"3852_CR10","doi-asserted-by":"publisher","first-page":"376","DOI":"10.1109\/34.88573","volume":"13","author":"S Umeyama","year":"1991","unstructured":"Umeyama, S.: Least-squares estimation of transformation parameters between two point patterns. IEEE Trans. Pattern Anal. Mach. Intell. 13(04), 376\u2013380 (1991)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"3852_CR11","doi-asserted-by":"crossref","unstructured":"Nguyen, V.N., Hu, Y., Xiao, Y., Salzmann, M., Lepetit, V.: Templates for 3d object pose estimation revisited: generalization to new objects and robustness to occlusions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6771\u20136780 (2022)","DOI":"10.1109\/CVPR52688.2022.00665"},{"key":"3852_CR12","doi-asserted-by":"crossref","unstructured":"Sundermeyer, M., Durner, M., Puang, E.Y., Marton, Z.-C., Vaskevicius, N., Arras, K.O., Triebel, R.: Multi-path learning for object pose estimation across domains. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13916\u201313925 (2020)","DOI":"10.1109\/CVPR42600.2020.01393"},{"key":"3852_CR13","doi-asserted-by":"crossref","unstructured":"Li, Y., Wang, G., Ji, X., Xiang, Y., Fox, D.: Deepim: deep iterative matching for 6d pose estimation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 683\u2013698 (2018)","DOI":"10.1007\/978-3-030-01231-1_42"},{"key":"3852_CR14","doi-asserted-by":"crossref","unstructured":"Liu, Y., Wen, Y., Peng, S., Lin, C., Long, X., Komura, T., Wang, W.: Gen6d: Generalizable model-free 6-dof object pose estimation from rgb images. In: European Conference on Computer Vision, pp. 298\u2013315. Springer (2022)","DOI":"10.1007\/978-3-031-19824-3_18"},{"key":"3852_CR15","first-page":"35103","volume":"35","author":"X He","year":"2022","unstructured":"He, X., Sun, J., Wang, Y., Huang, D., Bao, H., Zhou, X.: Onepose++: keypoint-free one-shot object pose estimation without cad models. Adv. Neural. Inf. Process. Syst. 35, 35103\u201335115 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"3852_CR16","doi-asserted-by":"crossref","unstructured":"Lin, J., Liu, L., Lu, D., Jia, K.: Sam-6d: segment anything model meets zero-shot 6d object pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 27906\u201327916 (2024)","DOI":"10.1109\/CVPR52733.2024.02636"},{"key":"3852_CR17","doi-asserted-by":"crossref","unstructured":"Wen, B., Yang, W., Kautz, J., Birchfield, S.: Foundationpose: unified 6d pose estimation and tracking of novel objects. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17868\u201317879 (2024)","DOI":"10.1109\/CVPR52733.2024.01692"},{"key":"3852_CR18","doi-asserted-by":"crossref","unstructured":"He, Y., Wang, Y., Fan, H., Sun, J., Chen, Q.: Fs6d: few-shot 6d pose estimation of novel objects. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6814\u20136824 (2022)","DOI":"10.1109\/CVPR52688.2022.00669"},{"key":"3852_CR19","doi-asserted-by":"crossref","unstructured":"Yen-Chen, L., Florence, P., Barron, J.T., Rodriguez, A., Isola, P., Lin, T.-Y.: inerf: inverting neural radiance fields for pose estimation. In: 2021 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 1323\u20131330. IEEE (2021)","DOI":"10.1109\/IROS51168.2021.9636708"},{"key":"3852_CR20","doi-asserted-by":"crossref","unstructured":"Li, F., Vutukur, S.R., Yu, H., Shugurov, I., Busam, B., Yang, S., Ilic, S.: Nerf-pose: a first-reconstruct-then-regress approach for weakly-supervised 6d object pose estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2123\u20132133 (2023)","DOI":"10.1109\/ICCVW60793.2023.00226"},{"key":"3852_CR21","unstructured":"Sun, Y., Wang, X., Zhang, Y., Zhang, J., Jiang, C., Guo, Y., Wang, F.: icomma: inverting 3d gaussians splatting for camera pose estimation via comparing and matching. arXiv preprint arXiv:2312.09031 (2023)"},{"key":"3852_CR22","doi-asserted-by":"crossref","unstructured":"Cai, D., Heikkil\u00e4, J., Rahtu, E.: Gs-pose: Cascaded framework for generalizable segmentation-based 6d object pose estimation. arXiv preprint arXiv:2403.10683 (2024)","DOI":"10.1109\/3DV66043.2025.00097"},{"key":"3852_CR23","doi-asserted-by":"crossref","unstructured":"Hinterstoisser, S., Lepetit, V., Ilic, S., Holzer, S., Bradski, G., Konolige, K., Navab, N.: Model based training, detection and pose estimation of texture-less 3d objects in heavily cluttered scenes. In: Computer Vision\u2013ACCV 2012: 11th Asian Conference on Computer Vision, Daejeon, Korea, November 5\u20139, 2012, Revised Selected Papers, Part I 11, pp. 548\u2013562. Springer (2013)","DOI":"10.1007\/978-3-642-37331-2_42"},{"key":"3852_CR24","doi-asserted-by":"crossref","unstructured":"Rad, M., Lepetit, V.: Bb8: a scalable, accurate, robust to partial occlusion method for predicting the 3d poses of challenging objects without using depth. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3828\u20133836 (2017)","DOI":"10.1109\/ICCV.2017.413"},{"key":"3852_CR25","doi-asserted-by":"crossref","unstructured":"Wei, L., Xie, F., Sun, L., Chen, J., Zhang, Z.: A modal fusion network with dual attention mechanism for 6d pose estimation. Vis. Comput. 40(10), 7411\u20137425 (2024)","DOI":"10.1007\/s00371-024-03614-w"},{"key":"3852_CR26","doi-asserted-by":"crossref","unstructured":"Chen, W., Duan, J., Basevi, H., Chang, H.J., Leonardis, A.: Pointposenet: point pose network for robust 6d object pose estimation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 2824\u20132833 (2020)","DOI":"10.1109\/WACV45572.2020.9093272"},{"key":"3852_CR27","doi-asserted-by":"publisher","first-page":"2011","DOI":"10.1007\/s00371-024-03520-1","volume":"41","author":"S Liu","year":"2025","unstructured":"Liu, S., Xu, F., Wu, C., Chi, J., Yu, X., Wei, L., Leng, C.: Cmt-6d: a lightweight iterative 6dof pose estimation network based on cross-modal transformer. Vis. Comput. 41, 2011\u20132027 (2025)","journal-title":"Vis. Comput."},{"key":"3852_CR28","doi-asserted-by":"crossref","unstructured":"Glasner, D., Galun, M., Alpert, S., Basri, R., Shakhnarovich, G.: Aware object detection and pose estimation. In: 2011 International Conference on Computer Vision, pp. 1275\u20131282. IEEE (2011)","DOI":"10.1109\/ICCV.2011.6126379"},{"issue":"8","key":"3852_CR29","doi-asserted-by":"publisher","first-page":"5421","DOI":"10.1007\/s00371-023-03113-4","volume":"40","author":"F Ullah","year":"2024","unstructured":"Ullah, F., Wei, W., Fan, Z., Yu, Q.: 6d object pose estimation based on dense convolutional object center voting with improved accuracy and efficiency. Vis. Comput. 40(8), 5421\u20135434 (2024)","journal-title":"Vis. Comput."},{"key":"3852_CR30","doi-asserted-by":"crossref","unstructured":"Peng, S., Liu, Y., Huang, Q., Zhou, X., Bao, H.: Pvnet: pixel-wise voting network for 6dof pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4561\u20134570 (2019)","DOI":"10.1109\/CVPR.2019.00469"},{"key":"3852_CR31","doi-asserted-by":"crossref","unstructured":"Wang, C., Xu, D., Zhu, Y., Mart\u00edn-Mart\u00edn, R., Lu, C., Fei-Fei, L., Savarese, S.: Densefusion: 6d object pose estimation by iterative dense fusion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3343\u20133352 (2019)","DOI":"10.1109\/CVPR.2019.00346"},{"key":"3852_CR32","doi-asserted-by":"crossref","unstructured":"Xiang, Y., Schmidt, T., Narayanan, V., Fox, D.: Posecnn: a convolutional neural network for 6d object pose estimation in cluttered scenes. arXiv preprint arXiv:1711.00199 (2017)","DOI":"10.15607\/RSS.2018.XIV.019"},{"key":"3852_CR33","doi-asserted-by":"crossref","unstructured":"Li, Z., Wang, G., Ji, X.: Cdpn: coordinates-based disentangled pose network for real-time rgb-based 6-dof object pose estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7678\u20137687 (2019)","DOI":"10.1109\/ICCV.2019.00777"},{"key":"3852_CR34","doi-asserted-by":"crossref","unstructured":"Zeng, A., Yu, K.-T., Song, S., Suo, D., Walker, E., Rodriguez, A., Xiao, J.: Multi-view self-supervised deep learning for 6d pose estimation in the amazon picking challenge. In: 2017 IEEE International Conference on Robotics and Automation (ICRA), pp. 1386\u20131383. IEEE (2017)","DOI":"10.1109\/ICRA.2017.7989165"},{"key":"3852_CR35","doi-asserted-by":"crossref","unstructured":"Hinterstoisser, S., Holzer, S., Cagniart, C., Ilic, S., Konolige, K., Navab, N., Lepetit, V.: Multimodal templates for real-time detection of texture-less objects in heavily cluttered scenes. In: 2011 International Conference on Computer Vision, pp. 858\u2013865. IEEE (2011)","DOI":"10.1109\/ICCV.2011.6126326"},{"key":"3852_CR36","doi-asserted-by":"crossref","unstructured":"Thalhammer, S., H\u00f6nig, P., Weibel, J.-B., Vincze, M.: Open challenges for monocular single-shot 6d object pose estimation. arXiv preprint arXiv:2302.11827 (2023)","DOI":"10.1109\/TRO.2024.3433870"},{"key":"3852_CR37","doi-asserted-by":"crossref","unstructured":"Peng, W., Yan, J., Wen, H., Sun, Y.: Self-supervised category-level 6d object pose estimation with deep implicit shape representation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 36, pp. 2082\u20132090 (2022)","DOI":"10.1609\/aaai.v36i2.20104"},{"key":"3852_CR38","doi-asserted-by":"crossref","unstructured":"Liu, J., Sun, W., Liu, C., Zhang, X., Fu, Q.: Robotic continuous grasping system by shape transformer-guided multi-object category-level 6d pose estimation. IEEE Trans. Ind. Inform. 19(11), 11171\u201311181 (2023)","DOI":"10.1109\/TII.2023.3244348"},{"key":"3852_CR39","doi-asserted-by":"crossref","unstructured":"Chen, K., James, S., Sui, C., Liu, Y.-H., Abbeel, P., Dou, Q.: Stereopose: category-level 6d transparent object pose estimation from stereo images via back-view nocs. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 2855\u20132861. IEEE (2023)","DOI":"10.1109\/ICRA48891.2023.10160780"},{"key":"3852_CR40","doi-asserted-by":"crossref","unstructured":"Zhang, H., Opipari, A., Chen, X., Zhu, J., Yu, Z., Jenkins, O.C.: Transnet: category-level transparent object pose estimation. In: European Conference on Computer Vision, pp. 148\u2013164. Springer (2022)","DOI":"10.1007\/978-3-031-25085-9_9"},{"key":"3852_CR41","unstructured":"He, Y., Fan, H., Huang, H., Chen, Q., Sun, J.: Towards self-supervised category-level object pose and size estimation. arXiv preprint arXiv:2203.02884 (2022)"},{"key":"3852_CR42","doi-asserted-by":"crossref","unstructured":"Tian, M., Ang, M.H., Lee, G.H.: Shape prior deformation for categorical 6d object pose and size estimation. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXI 16, pp. 530\u2013546. Springer (2020)","DOI":"10.1007\/978-3-030-58589-1_32"},{"key":"3852_CR43","doi-asserted-by":"crossref","unstructured":"Chen, W., Jia, X., Chang, H.J., Duan, J., Shen, L., Leonardis, A.: Fs-net: fast shape-based network for category-level 6d object pose estimation with decoupled rotation mechanism. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1581\u20131590 (2021)","DOI":"10.1109\/CVPR46437.2021.00163"},{"key":"3852_CR44","doi-asserted-by":"crossref","unstructured":"Park, K., Mousavian, A., Xiang, Y., Fox, D.: Latentfusion: end-to-end differentiable reconstruction and rendering for unseen object pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10710\u201310719 (2020)","DOI":"10.1109\/CVPR42600.2020.01072"},{"issue":"8","key":"3852_CR45","doi-asserted-by":"publisher","first-page":"2546","DOI":"10.1109\/TVCG.2019.2894627","volume":"26","author":"B Zhang","year":"2019","unstructured":"Zhang, B., Sheng, B., Li, P., Lee, T.-Y.: Depth of field rendering using multilayer-neighborhood optimization. IEEE Trans. Vis. Comput. Graphics 26(8), 2546\u20132559 (2019)","journal-title":"IEEE Trans. Vis. Comput. Graphics"},{"issue":"9","key":"3852_CR46","doi-asserted-by":"publisher","first-page":"1806","DOI":"10.1109\/TSMC.2018.2850149","volume":"49","author":"A Kamel","year":"2018","unstructured":"Kamel, A., Sheng, B., Yang, P., Li, P., Shen, R., Feng, D.D.: Deep convolutional neural networks for human action recognition using depth maps and postures. IEEE Trans. Syst. Man Cybern. Syst. 49(9), 1806\u20131819 (2018)","journal-title":"IEEE Trans. Syst. Man Cybern. Syst."},{"key":"3852_CR47","doi-asserted-by":"crossref","unstructured":"Sun, J., Wang, Z., Zhang, S., He, X., Zhao, H., Zhang, G., Zhou, X.: Onepose: one-shot object pose estimation without cad models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6825\u20136834 (2022)","DOI":"10.1109\/CVPR52688.2022.00670"},{"issue":"1","key":"3852_CR48","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1145\/3503250","volume":"65","author":"B Mildenhall","year":"2021","unstructured":"Mildenhall, B., Srinivasan, P.P., Tancik, M., Barron, J.T., Ramamoorthi, R., Ng, R.: Nerf: representing scenes as neural radiance fields for view synthesis. Commun. ACM 65(1), 99\u2013106 (2021)","journal-title":"Commun. ACM"},{"key":"3852_CR49","doi-asserted-by":"crossref","unstructured":"Kerbl, B., Kopanas, G., Leimk\u00fchler, T., Drettakis, G.: 3d gaussian splatting for real-time radiance field rendering. ACM Trans. Graphics 42(4), 139:1\u2013139:14 (2023)","DOI":"10.1145\/3592433"},{"key":"3852_CR50","doi-asserted-by":"crossref","unstructured":"Yen-Chen, L., Florence, P., Barron, J.T., Rodriguez, A., Isola, P., Lin, T.-Y.: inerf: inverting neural radiance fields for pose estimation. In: 2021 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 1323\u20131330. IEEE (2021)","DOI":"10.1109\/IROS51168.2021.9636708"},{"key":"3852_CR51","doi-asserted-by":"crossref","unstructured":"Li, F., Vutukur, S.R., Yu, H., Shugurov, I., Busam, B., Yang, S., Ilic, S.: Nerf-pose: a first-reconstruct-then-regress approach for weakly-supervised 6d object pose estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2123\u20132133 (2023)","DOI":"10.1109\/ICCVW60793.2023.00226"},{"key":"3852_CR52","doi-asserted-by":"crossref","unstructured":"Lin, Y., M\u00fcller, T., Tremblay, J., Wen, B., Tyree, S., Evans, A., Vela, P.A., Birchfield, S.: Parallel inversion of neural radiance fields for robust pose estimation. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 9377\u20139384. IEEE (2023)","DOI":"10.1109\/ICRA48891.2023.10161117"},{"key":"3852_CR53","unstructured":"Amir, S., Gandelsman, Y., Bagon, S., Dekel, T.: Deep vit features as dense visual descriptors. 2(3), 4 (2021). arXiv:2112.05814"},{"key":"3852_CR54","first-page":"15","volume":"36","author":"E Hedlin","year":"2024","unstructured":"Hedlin, E., Sharma, G., Mahajan, S., Isack, H., Kar, A., Tagliasacchi, A., Yi, K.M.: Unsupervised semantic correspondence using stable diffusion. Adv. Neural Inform. Process. Syst. 36, 15 (2024)","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"3852_CR55","doi-asserted-by":"crossref","unstructured":"Caron, M., Touvron, H., Misra, I., J\u00e9gou, H., Mairal, J., Bojanowski, P., Joulin, A.: Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9650\u20139660 (2021)","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"3852_CR56","unstructured":"Oquab, M., Darcet, T., Moutakanni, T., Vo, H., Szafraniec, M., Khalidov, V., Fernandez, P., Haziza, D., Massa, F., El-Nouby, A., et al.: Dinov2: learning robust visual features without supervision. arXiv preprint arXiv:2304.07193 (2023)"},{"key":"3852_CR57","doi-asserted-by":"crossref","unstructured":"Pan, P., Fan, Z., Feng, B.Y., Wang, P., Li, C., Wang, Z.: Learning to estimate 6dof pose from limited data: a few-shot, generalizable approach using rgb images. In: 2024 International Conference on 3D Vision (3DV), pp. 1059\u20131071. IEEE (2024)","DOI":"10.1109\/3DV62453.2024.00078"},{"key":"3852_CR58","doi-asserted-by":"crossref","unstructured":"Liu, S., Zeng, Z., Ren, T., Li, F., Zhang, H., Yang, J., Li, C., Yang, J., Su, H., Zhu, J., et al.: Grounding dino: marrying dino with grounded pre-training for open-set object detection. arXiv preprint arXiv:2303.05499 (2023)","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"3852_CR59","doi-asserted-by":"crossref","unstructured":"Jiang, H., Karpur, A., Cao, B., Huang, Q., Araujo, A.: Omniglue: generalizable feature matching with foundation model guidance. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19865\u201319875 (2024)","DOI":"10.1109\/CVPR52733.2024.01878"},{"key":"3852_CR60","doi-asserted-by":"crossref","unstructured":"Tumanyan, N., Singer, A., Bagon, S., Dekel, T.: Dino-tracker: Taming dino for self-supervised point tracking in a single video. arXiv preprint arXiv:2403.14548 (2024)","DOI":"10.1007\/978-3-031-73347-5_21"},{"key":"3852_CR61","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.-Y., et al.: Segment anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4015\u20134026 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"3852_CR62","doi-asserted-by":"crossref","unstructured":"Xie, Z., Guan, B., Jiang, W., Yi, M., Ding, Y., Lu, H., Zhang, L.: Pa-sam: Prompt adapter sam for high-quality image segmentation. arXiv preprint arXiv:2401.13051 (2024)","DOI":"10.1109\/ICME57554.2024.10687602"},{"key":"3852_CR63","unstructured":"Zhang, R., Jiang, Z., Guo, Z., Yan, S., Pan, J., Ma, X., Dong, H., Gao, P., Li, H.: Personalize segment anything model with one shot. arXiv preprint arXiv:2305.03048 (2023)"},{"key":"3852_CR64","doi-asserted-by":"crossref","unstructured":"Fan, Z., Pan, P., Wang, P., Jiang, Y., Xu, D., Wang, Z.: Pope: 6-dof promptable pose estimation of any object in any scene with one reference. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7771\u20137781 (2024)","DOI":"10.1109\/CVPRW63382.2024.00773"},{"key":"3852_CR65","doi-asserted-by":"crossref","unstructured":"Nguyen, V.N., Groueix, T., Ponimatkin, G., Lepetit, V., Hodan, T.: Cnos: a strong baseline for cad-based novel object segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2134\u20132140 (2023)","DOI":"10.1109\/ICCVW60793.2023.00227"},{"key":"3852_CR66","doi-asserted-by":"crossref","unstructured":"Pan, P., Fan, Z., Feng, B.Y., Wang, P., Li, C., Wang, Z.: Learning to estimate 6dof pose from limited data: a few-shot, generalizable approach using rgb images. In: 2024 International Conference on 3D Vision (3DV), pp. 1059\u20131071 (2024). IEEE","DOI":"10.1109\/3DV62453.2024.00078"},{"key":"3852_CR67","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"3852_CR68","doi-asserted-by":"crossref","unstructured":"Cai, D., Heikkil\u00e4, J., Rahtu, E.: Gs-pose: cascaded framework for generalizable segmentation-based 6d object pose estimation. arXiv preprint arXiv:2403.10683 (2024)","DOI":"10.1109\/3DV66043.2025.00097"},{"key":"3852_CR69","doi-asserted-by":"crossref","unstructured":"Shugurov, I., Li, F., Busam, B., Ilic, S.: Osop: a multi-stage one shot object pose estimation framework. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6835\u20136844 (2022)","DOI":"10.1109\/CVPR52688.2022.00671"},{"key":"3852_CR70","doi-asserted-by":"crossref","unstructured":"Wohlhart, P., Lepetit, V.: Learning descriptors for object recognition and 3d pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3109\u20133118 (2015)","DOI":"10.1109\/CVPR.2015.7298930"},{"key":"3852_CR71","unstructured":"Zhao, X., Ding, W., An, Y., Du, Y., Yu, T., Li, M., Tang, M., Wang, J.: Fast segment anything. arXiv preprint arXiv:2306.12156 (2023)"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03852-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-03852-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03852-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T07:31:02Z","timestamp":1757143862000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-03852-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,11]]},"references-count":71,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["3852"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-03852-6","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"type":"print","value":"0178-2789"},{"type":"electronic","value":"1432-2315"}],"subject":[],"published":{"date-parts":[[2025,3,11]]},"assertion":[{"value":"17 February 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 March 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"We rely on reference images with pose labels to achieve pose estimation for unseen objects, but the accuracy is still lower than instance-level methods. Additionally, our detector, selector, and refiner models are independently trained and executed. In the future, we aim to develop an end-to-end version and further reduce the reliance on reference view labels.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Limitation and discussions"}}]}}