{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,22]],"date-time":"2026-01-22T06:59:03Z","timestamp":1769065143970,"version":"3.49.0"},"reference-count":42,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62322607"],"award-info":[{"award-number":["62322607"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62236010"],"award-info":[{"award-number":["62236010"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62276261"],"award-info":[{"award-number":["62276261"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Beijing Natural Science Foundation","award":["L252033"],"award-info":[{"award-number":["L252033"]}]},{"name":"Taishan Scholar Foundation of Shandong Province","award":["tsqn202507043"],"award-info":[{"award-number":["tsqn202507043"]}]},{"DOI":"10.13039\/501100007129","name":"Natural Science Foundation of Shandong Province","doi-asserted-by":"publisher","award":["ZR2025QC1566"],"award-info":[{"award-number":["ZR2025QC1566"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Robot. Autom. Lett."],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1109\/lra.2026.3652073","type":"journal-article","created":{"date-parts":[[2026,1,12]],"date-time":"2026-01-12T23:58:44Z","timestamp":1768262324000},"page":"2482-2489","source":"Crossref","is-referenced-by-count":0,"title":["VERM: Leveraging Foundation Models to Create a\n                    <u>V<\/u>\n                    irtual\n                    <u>E<\/u>\n                    ye for Efficient 3D\n                    <u>R<\/u>\n                    obotic\n                    <u>M<\/u>\n                    anipulation"],"prefix":"10.1109","volume":"11","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-2895-9203","authenticated-orcid":false,"given":"Yixiang","family":"Chen","sequence":"first","affiliation":[{"name":"New Laboratory of Pattern Recognition (NLPR), State Key Laboratory of Multimodal Artificial Intelligence Systems (MAIS), Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8239-7229","authenticated-orcid":false,"given":"Yan","family":"Huang","sequence":"additional","affiliation":[{"name":"New Laboratory of Pattern Recognition (NLPR), State Key Laboratory of Multimodal Artificial Intelligence Systems (MAIS), Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5136-8444","authenticated-orcid":false,"given":"Keji","family":"He","sequence":"additional","affiliation":[{"name":"Shandong University, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-8404-5779","authenticated-orcid":false,"given":"Peiyan","family":"Li","sequence":"additional","affiliation":[{"name":"New Laboratory of Pattern Recognition (NLPR), State Key Laboratory of Multimodal Artificial Intelligence Systems (MAIS), Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5224-8647","authenticated-orcid":false,"given":"Liang","family":"Wang","sequence":"additional","affiliation":[{"name":"New Laboratory of Pattern Recognition (NLPR), State Key Laboratory of Multimodal Artificial Intelligence Systems (MAIS), Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2023.XIX.025"},{"key":"ref2","first-page":"2165","article-title":"RT-2: Vision-language-action models transfer web knowledge to robotic control","volume-title":"Proc. Conf. Robot Learn.","author":"Zitkovich","year":"2023"},{"key":"ref3","first-page":"785","article-title":"Perceiver-Actor: A multi-task transformer for robotic manipulation","volume-title":"Proc. Conf. Robot Learn.","author":"Shridhar","year":"2023"},{"key":"ref4","first-page":"694","article-title":"RVT: Robotic view transformer for 3D object manipulation","volume-title":"Proc. Conf. Robot Learn.","author":"Goyal","year":"2023"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2024.XX.055"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01337"},{"key":"ref7","first-page":"1761","article-title":"PolarNet: 3D point clouds for language-guided robotic manipulation","volume-title":"Proc. Conf. Robot Learn.","author":"Chen","year":"2023"},{"key":"ref8","first-page":"3949","article-title":"Act3D: 3D feature field transformers for multi-task robotic manipulation","volume-title":"Proc. 7th Annu. Conf. Robot Learn.","author":"Gervet","year":"2023"},{"key":"ref9","first-page":"1949","article-title":"3D diffuser actor: Policy diffusion with 3D scene representations","volume-title":"Proc. 8th Annu. Conf. Robot Learn.","author":"Ke","year":"2024"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2024.XX.067"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/IROS58592.2024.10802366"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1016\/j.tics.2005.02.009"},{"key":"ref13","article-title":"GPT-4o system card","author":"Hurst","year":"2024"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2024.3477090"},{"key":"ref15","article-title":"Look before you leap: Unveiling the power of GPT-4v in robotic vision-language planning","author":"Hu","year":"2023"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2974707"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1177\/02783649241281508"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2025.129963"},{"key":"ref19","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford","year":"2021"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20080-9_42"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160591"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3387941"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161317"},{"key":"ref26","first-page":"540","article-title":"VoxPoser: Composable 3D value maps for robotic manipulation with language models","volume-title":"Proc. Conf. Robot Learn.","author":"Huang","year":"2023"},{"key":"ref27","first-page":"4573","article-title":"ReKep: Spatio-temporal reasoning of relational keypoint constraints for robotic manipulation","volume-title":"Proc. 8th Annu. Conf. Robot Learn.","author":"Huang","year":"2024"},{"key":"ref28","article-title":"Video language planning","volume-title":"Proc. 12th Int. Conf. Learn. Representations","author":"Du","year":"2024"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2024.XX.106"},{"key":"ref30","first-page":"287","article-title":"Do as i can, not as i say: Grounding language in robotic affordances","volume-title":"Proc. Conf. Robot Learn.","author":"Brohan","year":"2023"},{"key":"ref31","first-page":"8469","article-title":"PaLM-E: An embodied multimodal language model","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Driess","year":"2023"},{"key":"ref32","first-page":"894","article-title":"CLIPort: What and where pathways for robotic manipulation","volume-title":"Proc. Conf. Robot Learn.","author":"Shridhar","year":"2022"},{"key":"ref33","article-title":"Set-of-mark prompting unleashes extraordinary visual grounding in GPT-4v","author":"Yang","year":"2023"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref35","first-page":"175","article-title":"Instruction-Driven history-aware policies for robotic manipulations","volume-title":"Proc. Conf. Robot Learn.","author":"Guhur","year":"2023"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2013.6696520"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1177\/0278364911406761"},{"key":"ref38","article-title":"Large batch optimization for deep learning: Training bert in 76 minutes","volume-title":"Proc. Int. Conf. Learn. Representations","author":"You","year":"2020"},{"key":"ref39","article-title":"Qwen2. 5 technical report","author":"Yang","year":"2024"},{"key":"ref40","article-title":"Claude 3.5 sonnet model card addendum","year":"2024"},{"key":"ref41","article-title":"VIOLA: Object-centric imitation learning for vision-based robot manipulation","volume-title":"Proc. 6th Annu. Conf. Robot Learn.","author":"Zhu","year":"2022"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2025.XXI.059"}],"container-title":["IEEE Robotics and Automation Letters"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7083369\/11359420\/11339964.pdf?arnumber=11339964","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,21]],"date-time":"2026-01-21T21:11:57Z","timestamp":1769029917000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11339964\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3]]},"references-count":42,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/lra.2026.3652073","relation":{},"ISSN":["2377-3766","2377-3774"],"issn-type":[{"value":"2377-3766","type":"electronic"},{"value":"2377-3774","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3]]}}}