{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,23]],"date-time":"2026-01-23T16:25:13Z","timestamp":1769185513938,"version":"3.49.0"},"publisher-location":"Cham","reference-count":56,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031925900","type":"print"},{"value":"9783031925917","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-92591-7_19","type":"book-chapter","created":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T07:24:23Z","timestamp":1747985063000},"page":"305-322","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Target-Oriented Object Grasping via\u00a0Multimodal Human Guidance"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1005-9252","authenticated-orcid":false,"given":"Pengwei","family":"Xie","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9235-6439","authenticated-orcid":false,"given":"Siang","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-7501-5504","authenticated-orcid":false,"given":"Yixiang","family":"Dai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2284-4824","authenticated-orcid":false,"given":"Dingchang","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0232-2698","authenticated-orcid":false,"given":"Kaiqin","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2131-3044","authenticated-orcid":false,"given":"Guijin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,12]]},"reference":[{"key":"19_CR1","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., et al.: Language models are few-shot learners. Adv. Neural. Inf. Process. Syst. 33, 1877\u20131901 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"19_CR2","doi-asserted-by":"crossref","unstructured":"Cao, Z., Simon, T., Wei, S.E., Sheikh, Y.: Realtime multi-person 2d pose estimation using part affinity fields. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7291\u20137299 (2017)","DOI":"10.1109\/CVPR.2017.143"},{"issue":"8","key":"19_CR3","doi-asserted-by":"publisher","first-page":"4895","DOI":"10.1109\/LRA.2023.3290513","volume":"8","author":"S Chen","year":"2023","unstructured":"Chen, S., Tang, W., Xie, P., Yang, W., Wang, G.: Efficient heatmap-guided 6-dof grasp detection in cluttered scenes. IEEE Robot. Autom. Lett. 8(8), 4895\u20134902 (2023). https:\/\/doi.org\/10.1109\/LRA.2023.3290513","journal-title":"IEEE Robot. Autom. Lett."},{"key":"19_CR4","doi-asserted-by":"crossref","unstructured":"Chen, Y., Xu, R., Lin, Y., Vela, P.A.: A joint network for grasp detection conditioned on natural language commands. In: 2021 IEEE International Conference on Robotics and Automation (ICRA), pp. 4576\u20134582 (2021)","DOI":"10.1109\/ICRA48506.2021.9561994"},{"issue":"1","key":"19_CR5","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1109\/MRA.2011.2181749","volume":"19","author":"S Chitta","year":"2012","unstructured":"Chitta, S., Sucan, I., Cousins, S.: Moveit![ros topics]. IEEE Robot. Autom. Mag. 19(1), 18\u201319 (2012)","journal-title":"IEEE Robot. Autom. Mag."},{"key":"19_CR6","doi-asserted-by":"crossref","unstructured":"Constantin, S., Eyiokur, F.I., Yaman, D., B\u00e4rmann, L., Waibel, A.: Interactive multimodal robot dialog using pointing gesture recognition. In: European Conference on Computer Vision, pp. 640\u2013657. Springer (2022)","DOI":"10.1007\/978-3-031-25075-0_43"},{"key":"19_CR7","doi-asserted-by":"crossref","unstructured":"Constantin, S., Eyiokur, F.I., Yaman, D., B\u00e4rmann, L., Waibel, A.: Multimodal error correction with natural language and pointing gestures. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1976\u20131986 (2023)","DOI":"10.1109\/ICCVW60793.2023.00212"},{"key":"19_CR8","doi-asserted-by":"crossref","unstructured":"Fang, H.S., et al.: Anygrasp: robust and efficient grasp perception in spatial and temporal domains. IEEE Trans. Robot. (2023)","DOI":"10.1109\/TRO.2023.3281153"},{"key":"19_CR9","doi-asserted-by":"crossref","unstructured":"Fang, H.S., Wang, C., Gou, M., Lu, C.: Graspnet-1billion: a large-scale benchmark for general object grasping. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11444\u201311453 (2020)","DOI":"10.1109\/CVPR42600.2020.01146"},{"key":"19_CR10","unstructured":"Gu, J., et\u00a0al.: Maniskill2: A unified benchmark for generalizable manipulation skills. arXiv preprint arXiv:2302.04659 (2023)"},{"issue":"11","key":"19_CR11","doi-asserted-by":"publisher","first-page":"1004","DOI":"10.1016\/j.robot.2008.08.012","volume":"56","author":"H Holzapfel","year":"2008","unstructured":"Holzapfel, H., Neubig, D., Waibel, A.: A dialogue approach to learning object descriptions and semantic categories. Robot. Auton. Syst. 56(11), 1004\u20131013 (2008)","journal-title":"Robot. Auton. Syst."},{"key":"19_CR12","volume-title":"OpenCV Computer Vision with Python","author":"J Howse","year":"2013","unstructured":"Howse, J.: OpenCV Computer Vision with Python, vol. 27. Packt Publishing Birmingham, UK (2013)"},{"key":"19_CR13","unstructured":"Kirillov, A., et\u00a0al.: Segment anything. arXiv preprint arXiv:2304.02643 (2023)"},{"key":"19_CR14","doi-asserted-by":"crossref","unstructured":"Kumra, S., Joshi, S., Sahin, F.: Antipodal robotic grasping using generative residual convolutional neural network. In: 2020 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS). IEEE (2020)","DOI":"10.1109\/IROS45743.2020.9340777"},{"key":"19_CR15","doi-asserted-by":"crossref","unstructured":"Liang, H., et al.: Pointnetgpd: detecting grasp configurations from point sets. In: 2019 International Conference on Robotics and Automation (ICRA). IEEE (2019)","DOI":"10.1109\/ICRA.2019.8794435"},{"key":"19_CR16","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2980\u20132988 (2017)","DOI":"10.1109\/ICCV.2017.324"},{"key":"19_CR17","doi-asserted-by":"crossref","unstructured":"Liu, P., Orru, Y., Paxton, C., Shafiullah, N.M.M., Pinto, L.: Ok-robot: What really matters in integrating open-knowledge models for robotics. arXiv preprint arXiv:2401.12202 (2024)","DOI":"10.15607\/RSS.2024.XX.091"},{"key":"19_CR18","doi-asserted-by":"crossref","unstructured":"Liu, S., et\u00a0al.: Grounding dino: Marrying dino with grounded pre-training for open-set object detection. arXiv preprint arXiv:2303.05499 (2023)","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"19_CR19","doi-asserted-by":"crossref","unstructured":"Liu, X., Zhang, Y., Cao, H., Shan, D., Zhao, J.: Joint segmentation and grasp pose detection with multi-modal feature fusion network. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 1751\u20131756. IEEE (2023)","DOI":"10.1109\/ICRA48891.2023.10160253"},{"key":"19_CR20","doi-asserted-by":"crossref","unstructured":"Liu, Z., Wang, Z., Huang, S., Zhou, J., Lu, J.: Ge-grasp: efficient target-oriented grasping in dense clutter. In: 2022 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 1388\u20131395. IEEE (2022)","DOI":"10.1109\/IROS47612.2022.9981499"},{"key":"19_CR21","doi-asserted-by":"crossref","unstructured":"Liu, Z., Chen, Z., Xie, S., Zheng, W.S.: Transgrasp: a multi-scale hierarchical point transformer for 7-dof grasp detection. In: 2022 International Conference on Robotics and Automation (ICRA). IEEE (2022)","DOI":"10.1109\/ICRA46639.2022.9812001"},{"key":"19_CR22","doi-asserted-by":"crossref","unstructured":"Lou, X., Yang, Y., Choi, C.: Collision-aware target-driven object grasping in constrained environments. In: 2021 IEEE International Conference on Robotics and Automation (ICRA), pp. 6364\u20136370 (2021)","DOI":"10.1109\/ICRA48506.2021.9561473"},{"key":"19_CR23","doi-asserted-by":"crossref","unstructured":"Lou, X., Yang, Y., Choi, C.: Learning object relations with graph neural networks for target-driven grasping in dense clutter. In: 2022 International Conference on Robotics and Automation (ICRA), pp. 742\u2013748 (2022)","DOI":"10.1109\/ICRA46639.2022.9811601"},{"key":"19_CR24","doi-asserted-by":"crossref","unstructured":"Lu, Y., et al.: Hybrid physical metric for 6-dof grasp pose detection. In: 2022 International Conference on Robotics and Automation (ICRA). IEEE (2022)","DOI":"10.1109\/ICRA46639.2022.9811961"},{"key":"19_CR25","doi-asserted-by":"crossref","unstructured":"Lu, Y., Fan, Y., Deng, B., Liu, F., Li, Y., Wang, S.: Vl-grasp: a 6-dof interactive grasp policy for language-oriented objects in cluttered indoor scenes. arXiv preprint arXiv:2308.00640 (2023)","DOI":"10.1109\/IROS55552.2023.10341379"},{"key":"19_CR26","unstructured":"Lugaresi, C., et\u00a0al.: Mediapipe: A framework for building perception pipelines. arXiv preprint arXiv:1906.08172 (2019)"},{"key":"19_CR27","unstructured":"Ma, H., Huang, D.: Towards scale balanced 6-dof grasp detection in cluttered scenes. In: Conference on Robot Learning, pp. 2004\u20132013. PMLR (2023)"},{"key":"19_CR28","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s40648-021-00200-w","volume":"8","author":"AC Medeiros","year":"2021","unstructured":"Medeiros, A.C., Ratsamee, P., Orlosky, J., Uranishi, Y., Higashida, M., Takemura, H.: 3d pointing gestures as target selection tools: guiding monocular UAVs during window selection in an outdoor environment. ROBOMECH J. 8, 1\u201319 (2021)","journal-title":"ROBOMECH J."},{"key":"19_CR29","unstructured":"Medeiros, L.: lang-segment-anything. https:\/\/github.com\/luca-medeiros\/lang-segment-anything (2023)"},{"issue":"3","key":"19_CR30","doi-asserted-by":"publisher","first-page":"131","DOI":"10.1007\/s43154-020-00015-4","volume":"1","author":"A Mohebbi","year":"2020","unstructured":"Mohebbi, A.: Human-robot interaction in rehabilitation and assistance: a review. Current Robot. Reports 1(3), 131\u2013144 (2020)","journal-title":"Current Robot. Reports"},{"key":"19_CR31","doi-asserted-by":"crossref","unstructured":"Morrison, D., Corke, P., Leitner, J.: Closing the loop for robotic grasping: A real-time, generative grasp synthesis approach. arXiv preprint arXiv:1804.05172 (2018)","DOI":"10.15607\/RSS.2018.XIV.021"},{"key":"19_CR32","doi-asserted-by":"crossref","unstructured":"Mousavian, A., Eppner, C., Fox, D.: 6-dof graspnet: variational grasp generation for object manipulation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2901\u20132910 (2019)","DOI":"10.1109\/ICCV.2019.00299"},{"key":"19_CR33","doi-asserted-by":"crossref","unstructured":"Murali, A., Mousavian, A., Eppner, C., Paxton, C., Fox, D.: 6-dof grasping for target-driven object manipulation in clutter. In: 2020 IEEE International Conference on Robotics and Automation (ICRA). IEEE (2020)","DOI":"10.1109\/ICRA40945.2020.9197318"},{"key":"19_CR34","doi-asserted-by":"crossref","unstructured":"Ni, P., Zhang, W., Zhu, X., Cao, Q.: Pointnet++ grasping:learning an end-to-end spatial grasp generation algorithm from sparse point clouds. In: 2020 IEEE International Conference on Robotics and Automation (ICRA). IEEE (2020)","DOI":"10.1109\/ICRA40945.2020.9196740"},{"issue":"13\u201314","key":"19_CR35","doi-asserted-by":"crossref","first-page":"1455","DOI":"10.1177\/0278364917735594","volume":"36","author":"A ten Pas","year":"2017","unstructured":"ten Pas, A., Gualtieri, M., Saenko, K., Platt, R.: Grasp pose detection in point clouds. Int. J. Robot. Res. 36(13\u201314), 1455\u20131473 (2017)","journal-title":"Int. J. Robot. Res."},{"key":"19_CR36","unstructured":"Qi, C.R., Su, H., Mo, K., Guibas, L.J.: Pointnet: deep learning on point sets for 3d classification and segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 652\u2013660 (2017)"},{"key":"19_CR37","unstructured":"Qi, C.R., Yi, L., Su, H., Guibas, L.J.: Pointnet++: deep hierarchical feature learning on point sets in a metric space. Advances in Neural Information Processing Systems 30 (2017)"},{"key":"19_CR38","unstructured":"Qin, Y., Chen, R., Zhu, H., Song, M., Xu, J., Su, H.: S4g: Amodal single-view single-shot se (3) grasp detection in cluttered scenes. In: Conference on robot learning. PMLR (2020)"},{"key":"19_CR39","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: Proceedings of the 38th International Conference on Machine Learning. vol.\u00a0139, pp. 8748\u20138763. PMLR (18\u201324 Jul 2021)"},{"key":"19_CR40","unstructured":"Ren, T., et al.: Grounded SAM: Assembling open-world models for diverse visual tasks (2024)"},{"issue":"1","key":"19_CR41","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3570731","volume":"12","author":"N Robinson","year":"2023","unstructured":"Robinson, N., Tidd, B., Campbell, D., Kuli\u0107, D., Corke, P.: Robotic vision for human-robot interaction and collaboration: a survey and systematic review. ACM Trans. Human-Robot Interact. 12(1), 1\u201366 (2023)","journal-title":"ACM Trans. Human-Robot Interact."},{"key":"19_CR42","doi-asserted-by":"publisher","DOI":"10.1016\/j.rcim.2022.102432","volume":"79","author":"F Semeraro","year":"2023","unstructured":"Semeraro, F., Griffiths, A., Cangelosi, A.: Human-robot collaboration and machine learning: a systematic review of recent research. Robot. Comput.-Integr. Manufact. 79, 102432 (2023)","journal-title":"Robot. Comput.-Integr. Manufact."},{"issue":"1","key":"19_CR43","first-page":"1","volume":"1","author":"O Sorkine-Hornung","year":"2017","unstructured":"Sorkine-Hornung, O., Rabinovich, M.: Least-squares rigid motion using svd. Computing 1(1), 1\u20135 (2017)","journal-title":"Computing"},{"key":"19_CR44","doi-asserted-by":"crossref","unstructured":"Sundermeyer, M., Mousavian, A., Triebel, R., Fox, D.: Contact-graspnet: Efficient 6-dof grasp generation in cluttered scenes. In: 2021 IEEE International Conference on Robotics and Automation (ICRA). IEEE (2021)","DOI":"10.1109\/ICRA48506.2021.9561877"},{"key":"19_CR45","unstructured":"Touvron, H., et\u00a0al.: Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)"},{"key":"19_CR46","doi-asserted-by":"crossref","unstructured":"Wang, C., Fang, H.S., Gou, M., Fang, H., Gao, J., Lu, C.: Graspness discovery in clutters for fast and accurate grasp detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15964\u201315973 (2021)","DOI":"10.1109\/ICCV48922.2021.01566"},{"key":"19_CR47","doi-asserted-by":"crossref","unstructured":"Wei, W., Luo, Y., Li, F., Xu, G., Zhong, J., Li, W., Wang, P.: Gpr: grasp pose refinement network for cluttered scenes. In: 2021 IEEE International Conference on Robotics and Automation (ICRA). IEEE (2021)","DOI":"10.1109\/ICRA48506.2021.9561868"},{"issue":"8","key":"19_CR48","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3547138","volume":"55","author":"H Weld","year":"2022","unstructured":"Weld, H., Huang, X., Long, S., Poon, J., Han, S.C.: A survey of joint intent detection and slot filling models in natural language understanding. ACM Comput. Surv. 55(8), 1\u201338 (2022)","journal-title":"ACM Comput. Surv."},{"key":"19_CR49","doi-asserted-by":"crossref","unstructured":"Xu, K., et al.: A joint modeling of vision-language-action for target-oriented grasping in clutter. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 11597\u201311604 (2023)","DOI":"10.1109\/ICRA48891.2023.10161041"},{"key":"19_CR50","doi-asserted-by":"crossref","unstructured":"Xu, Z., Xu, K., Xiong, R., Wang, Y.: Object-centric inference for language conditioned placement: a foundation model based approach. In: 2023 International Conference on Advanced Robotics and Mechatronics (ICARM), pp. 203\u2013208. IEEE (2023)","DOI":"10.1109\/ICARM58088.2023.10218865"},{"key":"19_CR51","unstructured":"Yang, J., et al.: Transferring foundation models for generalizable robotic manipulation. arXiv e-prints arXiv:2306 (2023)"},{"key":"19_CR52","doi-asserted-by":"crossref","unstructured":"Yang, Z., Sun, Y., Liu, S., Qi, X., Jia, J.: Cn: Channel normalization for point cloud recognition. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part X 16, pp. 600\u2013616. Springer (2020)","DOI":"10.1007\/978-3-030-58607-2_35"},{"key":"19_CR53","unstructured":"Yenamandra, S., et\u00a0al.: Homerobot: Open-vocabulary mobile manipulation. arXiv preprint arXiv:2306.11565 (2023)"},{"key":"19_CR54","unstructured":"Zhang, C., et al.: Faster segment anything: Towards lightweight sam for mobile applications. arXiv preprint arXiv:2306.14289 (2023)"},{"key":"19_CR55","doi-asserted-by":"crossref","unstructured":"Zhao, B., Zhang, H., Lan, X., Wang, H., Tian, Z., Zheng, N.: Regnet: region-based grasp network for end-to-end grasp detection in point clouds. In: 2021 IEEE International Conference on Robotics and Automation (ICRA). IEEE (2021)","DOI":"10.1109\/ICRA48506.2021.9561920"},{"key":"19_CR56","unstructured":"Zhao, X., et al.: Fast segment anything. arXiv preprint arXiv:2306.12156 (2023)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-92591-7_19","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T07:24:53Z","timestamp":1747985093000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-92591-7_19"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031925900","9783031925917"],"references-count":56,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-92591-7_19","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"12 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}