{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T15:59:14Z","timestamp":1780675154873,"version":"3.54.1"},"reference-count":49,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,5,29]],"date-time":"2023-05-29T00:00:00Z","timestamp":1685318400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,5,29]],"date-time":"2023-05-29T00:00:00Z","timestamp":1685318400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,5,29]]},"DOI":"10.1109\/icra48891.2023.10161041","type":"proceedings-article","created":{"date-parts":[[2023,7,4]],"date-time":"2023-07-04T17:20:56Z","timestamp":1688491256000},"page":"11597-11604","source":"Crossref","is-referenced-by-count":52,"title":["A Joint Modeling of Vision-Language-Action for Target-oriented Grasping in Clutter"],"prefix":"10.1109","author":[{"given":"Kechun","family":"Xu","sequence":"first","affiliation":[{"name":"Zhejiang University,Hangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuqi","family":"Zhao","sequence":"additional","affiliation":[{"name":"Zhejiang University,Hangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhongxiang","family":"Zhou","sequence":"additional","affiliation":[{"name":"Zhejiang University,Hangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zizhang","family":"Li","sequence":"additional","affiliation":[{"name":"Zhejiang University,Hangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huaijin","family":"Pi","sequence":"additional","affiliation":[{"name":"Zhejiang University,Hangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yifeng","family":"Zhu","sequence":"additional","affiliation":[{"name":"University of Texas at Austin,United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yue","family":"Wang","sequence":"additional","affiliation":[{"name":"Zhejiang University,Hangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rong","family":"Xiong","sequence":"additional","affiliation":[{"name":"Zhejiang University,Hangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref13","article-title":"Vlmbench: A compositional benchmark for vision-and-language manipulation","author":"zheng","year":"20","journal-title":"ArXiv Preprint"},{"key":"ref12","first-page":"894","article-title":"Cliport: What and where pathways for robotic manipulation","author":"shridhar","year":"0","journal-title":"Conference on Robot Learning"},{"key":"ref15","article-title":"Interactive visual grounding of re-ferring expressions for human-robot interaction","author":"shridhar","year":"20","journal-title":"ArXiv Preprint"},{"key":"ref14","first-page":"3774","article-title":"Interactively picking real-world objects with un-constrained spoken language instructions","author":"hatori","year":"0","journal-title":"2018 IEEE International Conference on Robotics and Automation (ICRA)"},{"key":"ref11","article-title":"Inner monologue: Embodied reasoning through planning with language models","author":"huang","year":"20","journal-title":"ArXiv Preprint"},{"key":"ref10","article-title":"Do as i can, not as i say: Grounding language in robotic affordances","author":"ahn","year":"20","journal-title":"ArXiv Preprint"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2021.XVII.020"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9812360"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01146"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9811817"},{"key":"ref46","first-page":"2016","article-title":"Pybullet, a python module for physics sim-ulation for games, robotics and machine learning","author":"coumans","year":"0"},{"key":"ref45","article-title":"Paper with appendix","author":"xu","year":"2021"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3129136"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref42","first-page":"405","article-title":"Nerf: Representing scenes as neural radiance fields for view synthesis","author":"mildenhall","year":"2020","journal-title":"Computer Vision-ECCV 2020 16th European Conference"},{"key":"ref41","article-title":"Attention is all you need","volume":"30","author":"vaswani","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref44","article-title":"Soft actor-critic for discrete action settings","author":"christodoulou","year":"20","journal-title":"ArXiv Preprint"},{"key":"ref43","article-title":"Soft actor-critic algorithms and applications","author":"haarnoja","year":"20","journal-title":"ArXiv Preprint"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11671"},{"key":"ref8","first-page":"1877","article-title":"Language mod-els are few-shot learners","volume":"33","author":"brown","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref7","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"devlin","year":"20","journal-title":"ar Xiv preprint"},{"key":"ref9","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"radford","year":"0","journal-title":"International Conference on Machine Learning"},{"key":"ref4","first-page":"119","article-title":"End-to-end learning of semantic grasping","author":"jang","year":"0","journal-title":"Conference on Robot Learning"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8461041"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/COASE.2016.7743488"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3131378"},{"key":"ref40","first-page":"69","article-title":"Modeling con-text in referring expressions","author":"yu","year":"2016","journal-title":"European Conference on Computer Vision"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9811815"},{"key":"ref34","first-page":"882","article-title":"Learning robotic grasping strat-egy based on natural-language object descriptions","author":"rao","year":"0","journal-title":"2018 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1177\/0278364915602060"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561994"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3092640"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2970622"},{"key":"ref33","first-page":"13 139","article-title":"Language-conditioned imitation learning for robot ma-nipulation tasks","volume":"33","author":"stepputtis","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3123373"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9197318"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1177\/0278364919868017"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2018.2852786"},{"key":"ref38","first-page":"1486","article-title":"Guiding multi-step rearrangement tasks with natural language instructions","author":"stengel-eskin","year":"0","journal-title":"Conference on Robot Learning"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2017.XIII.058"},{"key":"ref23","first-page":"307","article-title":"Using geometry to detect grasp poses in 3d point clouds","author":"pas","year":"2018","journal-title":"Robotics Research"},{"key":"ref26","first-page":"15964","article-title":"Grasp-ness discovery in clutters for fast and accurate grasp detection","author":"wang","year":"0","journal-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561877"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2016.7487517"},{"key":"ref22","first-page":"651","article-title":"Scalable deep reinforcement learning for vision-based robotic manipulation","author":"kalashnikov","year":"0","journal-title":"Conference on Robot Learning"},{"key":"ref21","first-page":"515","article-title":"Learning deep policies for robot bin picking by simulating robust grasping sequences","author":"mahler","year":"0","journal-title":"Conference on Robot Learning"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8793972"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3181735"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/IROS45743.2020.9341545"}],"event":{"name":"2023 IEEE International Conference on Robotics and Automation (ICRA)","location":"London, United Kingdom","start":{"date-parts":[[2023,5,29]]},"end":{"date-parts":[[2023,6,2]]}},"container-title":["2023 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10160211\/10160212\/10161041.pdf?arnumber=10161041","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,24]],"date-time":"2025-02-24T18:34:04Z","timestamp":1740422044000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10161041\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,5,29]]},"references-count":49,"URL":"https:\/\/doi.org\/10.1109\/icra48891.2023.10161041","relation":{},"subject":[],"published":{"date-parts":[[2023,5,29]]}}}