{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T11:40:21Z","timestamp":1782301221339,"version":"3.54.5"},"publisher-location":"Cham","reference-count":47,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031726972","type":"print"},{"value":"9783031726989","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T00:00:00Z","timestamp":1729900800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T00:00:00Z","timestamp":1729900800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72698-9_13","type":"book-chapter","created":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T04:45:57Z","timestamp":1729831557000},"page":"216-232","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["Omni6D: Large-Vocabulary 3D Object Dataset for\u00a0Category-Level 6D Object Pose Estimation"],"prefix":"10.1007","author":[{"given":"Mengchen","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tong","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tai","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tengfei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ziwei","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dahua","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,26]]},"reference":[{"key":"13_CR1","unstructured":"Barrow, H.G., Tenenbaum, J.M., Bolles, R.C., Wolf, H.C.: Parametric correspondence and chamfer matching: two new techniques for image matching. In: IJCAI, pp. 659\u2013663. William Kaufmann (1977)"},{"issue":"2","key":"13_CR2","doi-asserted-by":"publisher","first-page":"239","DOI":"10.1109\/34.121791","volume":"14","author":"PJ Besl","year":"1992","unstructured":"Besl, P.J., McKay, N.D.: A method for registration of 3-D shapes. IEEE Trans. Pattern Anal. Mach. Intell. 14(2), 239\u2013256 (1992)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"13_CR3","doi-asserted-by":"crossref","unstructured":"Brachmann, E., Michel, F., Krull, A., Yang, M.Y., Gumhold, S., Rother, C.: Uncertainty-driven 6D pose estimation of objects and scenes from a single RGB image. In: CVPR, pp. 3364\u20133372 (2016)","DOI":"10.1109\/CVPR.2016.366"},{"key":"13_CR4","doi-asserted-by":"crossref","unstructured":"Brazil, G., Kumar, A., Straub, J., Ravi, N., Johnson, J., Gkioxari, G.: Omni3D: a large benchmark and model for 3D object detection in the wild. In: CVPR, pp. 13154\u201313164 (2023)","DOI":"10.1109\/CVPR52729.2023.01264"},{"key":"13_CR5","doi-asserted-by":"crossref","unstructured":"Chen, D., Li, J., Wang, Z., Xu, K.: Learning canonical shape space for category-level 6D object pose and size estimation. In: CVPR, pp. 11970\u201311979. Computer Vision Foundation \/ IEEE (2020)","DOI":"10.1109\/CVPR42600.2020.01199"},{"key":"13_CR6","doi-asserted-by":"crossref","unstructured":"Chen, K., Dou, Q.: SGPA: structure-guided prior adaptation for category-level 6D object pose estimation. In: ICCV, pp. 2753\u20132762 (2021)","DOI":"10.1109\/ICCV48922.2021.00277"},{"key":"13_CR7","doi-asserted-by":"crossref","unstructured":"Chen, W., Jia, X., Chang, H.J., Duan, J., Shen, L., Leonardis, A.: FS-Net: fast shape-based network for category-level 6D object pose estimation with decoupled rotation mechanism. In: CVPR, pp. 1581\u20131590 (2021)","DOI":"10.1109\/CVPR46437.2021.00163"},{"key":"13_CR8","doi-asserted-by":"crossref","unstructured":"Chen, X., Dong, Z., Song, J., Geiger, A., Hilliges, O.: Category level object pose estimation via neural analysis-by-synthesis. In: ECCV (26), pp. 139\u2013156 (2020)","DOI":"10.1007\/978-3-030-58574-7_9"},{"issue":"83","key":"13_CR9","doi-asserted-by":"publisher","first-page":"4901","DOI":"10.21105\/joss.04901","volume":"8","author":"M Denninger","year":"2023","unstructured":"Denninger, M., et al.: BlenderProc2: a procedural pipeline for photorealistic rendering. J. Open Source Softw. 8(83), 4901 (2023)","journal-title":"J. Open Source Softw."},{"key":"13_CR10","doi-asserted-by":"crossref","unstructured":"Di, Y., et al.: GPV-Pose: category-level object pose estimation via geometry-guided point-wise voting. In: CVPR, pp. 6771\u20136781 (2022)","DOI":"10.1109\/CVPR52688.2022.00666"},{"issue":"3","key":"13_CR11","doi-asserted-by":"publisher","first-page":"1677","DOI":"10.1007\/s10462-020-09888-5","volume":"54","author":"G Du","year":"2021","unstructured":"Du, G., Wang, K., Lian, S., Zhao, K.: Vision-based robotic grasping from object localization, object pose estimation to grasp estimation for parallel grippers: a review. Artif. Intell. Rev. 54(3), 1677\u20131734 (2021)","journal-title":"Artif. Intell. Rev."},{"issue":"6","key":"13_CR12","doi-asserted-by":"publisher","first-page":"381","DOI":"10.1145\/358669.358692","volume":"24","author":"MA Fischler","year":"1981","unstructured":"Fischler, M.A., Bolles, R.C.: Random sample consensus: a paradigm for model fitting with applications to image analysis and automated cartography. Commun. ACM 24(6), 381\u2013395 (1981)","journal-title":"Commun. ACM"},{"key":"13_CR13","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-16-4939-4","volume-title":"Introduction to Visual SLAM","author":"X Gao","year":"2021","unstructured":"Gao, X., Zhang, T.: Introduction to Visual SLAM. Springer, Singapore (2021). https:\/\/doi.org\/10.1007\/978-981-16-4939-4"},{"issue":"11","key":"13_CR14","doi-asserted-by":"publisher","first-page":"1231","DOI":"10.1177\/0278364913491297","volume":"32","author":"A Geiger","year":"2013","unstructured":"Geiger, A., Lenz, P., Stiller, C., Urtasun, R.: Vision meets robotics: the KITTI dataset. Int. J. Robotics Res. 32(11), 1231\u20131237 (2013)","journal-title":"Int. J. Robotics Res."},{"key":"13_CR15","unstructured":"Huang, S., Qi, S., Xiao, Y., Zhu, Y., Wu, Y.N., Zhu, S.: Cooperative holistic scene understanding: unifying 3D object, layout, and camera pose estimation. In: NeurIPS, pp. 206\u2013217 (2018)"},{"key":"13_CR16","doi-asserted-by":"crossref","unstructured":"Irshad, M.Z., Kollar, T., Laskey, M., Stone, K., Kira, Z.: CenterSnap: single-shot multi-object 3D shape reconstruction and categorical 6D pose and size estimation. In: ICRA, pp. 10632\u201310640. IEEE (2022)","DOI":"10.1109\/ICRA46639.2022.9811799"},{"key":"13_CR17","doi-asserted-by":"publisher","unstructured":"Irshad, M.Z., Zakharov, S., Ambrus, R., Kollar, T., Kira, Z., Gaidon, A.: ShAPO: implicit representations for multi-object shape, appearance, and pose optimization. In: ECCV (2). Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-20086-1_16","DOI":"10.1007\/978-3-031-20086-1_16"},{"key":"13_CR18","doi-asserted-by":"crossref","unstructured":"Kirillov, A., et al.: Segment anything. arXiv:2304.02643 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"issue":"3","key":"13_CR19","doi-asserted-by":"publisher","first-page":"657","DOI":"10.1007\/s11263-019-01250-9","volume":"128","author":"Y Li","year":"2020","unstructured":"Li, Y., Wang, G., Ji, X., Xiang, Y., Fox, D.: DeepIM: deep iterative matching for 6D pose estimation. Int. J. Comput. Vis. 128(3), 657\u2013678 (2020)","journal-title":"Int. J. Comput. Vis."},{"key":"13_CR20","doi-asserted-by":"crossref","unstructured":"Lin, J., Wei, Z., Li, Z., Xu, S., Jia, K., Li, Y.: DualPoseNet: category-level 6D object pose and size estimation using dual pose network with refined learning of pose consistency. In: ICCV, pp. 3540\u20133549 (2021)","DOI":"10.1109\/ICCV48922.2021.00354"},{"key":"13_CR21","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., et al.: Microsoft COCO: common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13, pp. 740\u2013755 (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"13_CR22","unstructured":"Lin, Y., Florence, P., Barron, J.T., Rodriguez, A., Isola, P., Lin, T.: INeRF: inverting neural radiance fields for pose estimation. In: IROS, pp. 1323\u20131330 (2021)"},{"key":"13_CR23","doi-asserted-by":"crossref","unstructured":"Liu, J., Chen, Y., Ye, X., Qi, X.: Prior-free category-level pose estimation with implicit space transformation. CoRR abs\/2303.13479 (2023)","DOI":"10.1109\/ICCV51070.2023.01285"},{"key":"13_CR24","doi-asserted-by":"crossref","unstructured":"Liu, X., Wang, G., Li, Y., Ji, X.: CATRE: iterative point clouds alignment for category-level object pose refinement. In: ECCV (2), pp. 499\u2013516 (2022)","DOI":"10.1007\/978-3-031-20086-1_29"},{"key":"13_CR25","doi-asserted-by":"crossref","unstructured":"Lunayach, M., Zakharov, S., Chen, D., Ambrus, R., Kira, Z., Irshad, M.Z.: FSD: fast self-supervised single RGB-D to categorical 3D objects. CoRR abs\/2310.12974 (2023)","DOI":"10.1109\/ICRA57147.2024.10611012"},{"issue":"12","key":"13_CR26","doi-asserted-by":"publisher","first-page":"2633","DOI":"10.1109\/TVCG.2015.2513408","volume":"22","author":"\u00c9 Marchand","year":"2016","unstructured":"Marchand, \u00c9., Uchiyama, H., Spindler, F.: Pose estimation for augmented reality: a hands-on survey. IEEE Trans. Vis. Comput. Graph. 22(12), 2633\u20132651 (2016)","journal-title":"IEEE Trans. Vis. Comput. Graph."},{"issue":"3","key":"13_CR27","doi-asserted-by":"publisher","first-page":"274","DOI":"10.1007\/s00357-014-9161-z","volume":"31","author":"F Murtagh","year":"2014","unstructured":"Murtagh, F., Legendre, P.: Ward\u2019s hierarchical agglomerative clustering method: which algorithms implement ward\u2019s criterion? J. Classif. 31(3), 274\u2013295 (2014)","journal-title":"J. Classif."},{"key":"13_CR28","doi-asserted-by":"crossref","unstructured":"Nie, Y., Han, X., Guo, S., Zheng, Y., Chang, J., Zhang, J.: Total3Dunderstanding: joint layout, object pose and mesh reconstruction for indoor scenes from a single image. In: CVPR, pp. 52\u201361 (2020)","DOI":"10.1109\/CVPR42600.2020.00013"},{"key":"13_CR29","doi-asserted-by":"crossref","unstructured":"Peng, W., Yan, J., Wen, H., Sun, Y.: Self-supervised category-level 6d object pose estimation with deep implicit shape representation. In: AAAI, pp. 2082\u20132090. AAAI Press (2022)","DOI":"10.1609\/aaai.v36i2.20104"},{"key":"13_CR30","doi-asserted-by":"crossref","unstructured":"Rad, M., Lepetit, V.: BB8: a scalable, accurate, robust to partial occlusion method for predicting the 3D poses of challenging objects without using depth. In: ICCV, pp. 3848\u20133856 (2017)","DOI":"10.1109\/ICCV.2017.413"},{"key":"13_CR31","doi-asserted-by":"crossref","unstructured":"Shotton, J., Glocker, B., Zach, C., Izadi, S., Criminisi, A., Fitzgibbon, A.W.: Scene coordinate regression forests for camera relocalization in RGB-D images. In: CVPR, pp. 2930\u20132937 (2013)","DOI":"10.1109\/CVPR.2013.377"},{"key":"13_CR32","doi-asserted-by":"crossref","unstructured":"Song, C., Song, J., Huang, Q.: HybridPose: 6D object pose estimation under hybrid representations. In: CVPR, pp. 428\u2013437 (2020)","DOI":"10.1109\/CVPR42600.2020.00051"},{"key":"13_CR33","doi-asserted-by":"crossref","unstructured":"Su, Y., Rambach, J.R., Minaskan, N., Lesur, P., Pagani, A., Stricker, D.: Deep multi-state object pose estimation for augmented reality assembly. In: ISMAR Adjunct, pp. 222\u2013227. IEEE (2019)","DOI":"10.1109\/ISMAR-Adjunct.2019.00-42"},{"key":"13_CR34","doi-asserted-by":"crossref","unstructured":"Tian, M., Ang, M.H., Lee, G.H.: Shape prior deformation for categorical 6d object pose and size estimation. In: ECCV (21), pp. 530\u2013546 (2020)","DOI":"10.1007\/978-3-030-58589-1_32"},{"key":"13_CR35","unstructured":"Tremblay, J., To, T., Sundaralingam, B., Xiang, Y., Fox, D., Birchfield, S.: Deep object pose estimation for semantic robotic grasping of household objects. In: CoRL, pp. 306\u2013316 (2018)"},{"issue":"4","key":"13_CR36","doi-asserted-by":"publisher","first-page":"376","DOI":"10.1109\/34.88573","volume":"13","author":"S Umeyama","year":"1991","unstructured":"Umeyama, S.: Least-squares estimation of transformation parameters between two point patterns. IEEE Trans. Pattern Anal. Mach. Intell. 13(4), 376\u2013380 (1991)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"13_CR37","doi-asserted-by":"crossref","unstructured":"Wang, C., et al.: 6-pack: category-level 6d pose tracker with anchor-based keypoints. In: ICRA, pp. 10059\u201310066 (2020)","DOI":"10.1109\/ICRA40945.2020.9196679"},{"key":"13_CR38","unstructured":"Wang, G., Manhardt, F., Liu, X., Ji, X., Tombari, F.: Occlusion-aware self-supervised monocular 6D object pose estimation. CoRR abs\/2203.10339 (2022)"},{"key":"13_CR39","doi-asserted-by":"crossref","unstructured":"Wang, G., Manhardt, F., Tombari, F., Ji, X.: GDR-Net: geometry-guided direct regression network for monocular 6D object pose estimation. In: CVPR, pp. 16611\u201316621 (2021)","DOI":"10.1109\/CVPR46437.2021.01634"},{"key":"13_CR40","doi-asserted-by":"crossref","unstructured":"Wang, H., Sridhar, S., Huang, J., Valentin, J., Song, S., Guibas, L.J.: Normalized object coordinate space for category-level 6D object pose and size estimation. In: CVPR, pp. 2642\u20132651 (2019)","DOI":"10.1109\/CVPR.2019.00275"},{"key":"13_CR41","doi-asserted-by":"crossref","unstructured":"Wu, T., et al.: OmniObject3D: large-vocabulary 3D object dataset for realistic perception, reconstruction and generation. In: CVPR, pp. 803\u2013814 (2023)","DOI":"10.1109\/CVPR52729.2023.00084"},{"key":"13_CR42","doi-asserted-by":"crossref","unstructured":"Xiang, Y., Schmidt, T., Narayanan, V., Fox, D.: PoseCNN: a convolutional neural network for 6D object pose estimation in cluttered scenes. In: Robotics: Science and Systems (2018)","DOI":"10.15607\/RSS.2018.XIV.019"},{"key":"13_CR43","doi-asserted-by":"crossref","unstructured":"Zakharov, S., Shugurov, I., Ilic, S.: DPOD: dense 6D pose object detector in RGB images. CoRR abs\/1902.11020 (2019)","DOI":"10.1109\/ICCV.2019.00203"},{"key":"13_CR44","unstructured":"Ze, Y., Wang, X.: Category-level 6D object pose estimation in the wild: a semi-supervised learning approach and a new dataset. In: NeurIPS (2022)"},{"key":"13_CR45","unstructured":"Zhang, K., Fu, Y., Borse, S., Cai, H., Porikli, F., Wang, X.: Self-supervised geometric correspondence for category-level 6D object pose estimation in the wild. In: ICLR. OpenReview.net (2023)"},{"key":"13_CR46","doi-asserted-by":"crossref","unstructured":"Zhang, R., Di, Y., Lou, Z., Manhardt, F., Tombari, F., Ji, X.: RBP-Pose: residual bounding box projection for category-level pose estimation. In: ECCV (1), pp. 655\u2013672 (2022)","DOI":"10.1007\/978-3-031-19769-7_38"},{"key":"13_CR47","doi-asserted-by":"crossref","unstructured":"Zheng, L., et al.: Hs-pose: hybrid scope feature extraction for category-level object pose estimation. In: CVPR, pp. 17163\u201317173 (2023)","DOI":"10.1109\/CVPR52729.2023.01646"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72698-9_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T04:49:54Z","timestamp":1729831794000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72698-9_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,26]]},"ISBN":["9783031726972","9783031726989"],"references-count":47,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72698-9_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,26]]},"assertion":[{"value":"26 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}