{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,22]],"date-time":"2026-01-22T00:32:27Z","timestamp":1769041947279,"version":"3.49.0"},"reference-count":52,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,10,14]],"date-time":"2024-10-14T00:00:00Z","timestamp":1728864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,10,14]],"date-time":"2024-10-14T00:00:00Z","timestamp":1728864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,10,14]]},"DOI":"10.1109\/iros58592.2024.10802366","type":"proceedings-article","created":{"date-parts":[[2024,12,25]],"date-time":"2024-12-25T19:17:39Z","timestamp":1735154259000},"page":"403-410","source":"Crossref","is-referenced-by-count":3,"title":["VIHE: Virtual In-Hand Eye Transformer for 3D Robotic Manipulation"],"prefix":"10.1109","author":[{"given":"Weiyao","family":"Wang","sequence":"first","affiliation":[{"name":"Johns Hopkins University,Department of Computer Science,Baltimore,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yutian","family":"Lei","sequence":"additional","affiliation":[{"name":"Baidu Research, Robotics and Autonomous Driving Lab (RAL),Sunnyvale,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shiyu","family":"Jin","sequence":"additional","affiliation":[{"name":"Baidu Research, Robotics and Autonomous Driving Lab (RAL),Sunnyvale,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gregory D.","family":"Hager","sequence":"additional","affiliation":[{"name":"Johns Hopkins University,Department of Computer Science,Baltimore,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liangjun","family":"Zhang","sequence":"additional","affiliation":[{"name":"Baidu Research, Robotics and Autonomous Driving Lab (RAL),Sunnyvale,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2008.10.024"},{"issue":"30","key":"ref2","first-page":"1","article-title":"A review of robot learning for manipulation: Challenges, representations, and algorithms","volume":"22","author":"Kroemer","year":"2021","journal-title":"Journal of machine learning research"},{"key":"ref3","article-title":"Do as i can, not as i say: Grounding language in robotic affordances","author":"Ahn","year":"2022"},{"key":"ref4","first-page":"9118","article-title":"Language models as zero-shot planners: Extracting actionable knowledge for embodied agents","volume-title":"International Conference on Machine Learning","author":"Huang"},{"key":"ref5","article-title":"Palm-e: An embodied multimodal language model","author":"Driess","year":"2023"},{"key":"ref6","first-page":"785","article-title":"Perceiver-actor: A multi-task transformer for robotic manipulation","volume-title":"Conference on Robot Learning","author":"Shridhar"},{"key":"ref7","article-title":"Rvt: Robotic view transformer for 3d object manipulation","author":"Goyal","year":"2023"},{"key":"ref8","article-title":"Act3d: Infinite resolution action detection transformer for robotic manipulation","author":"Gervet","year":"2023"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3144512"},{"key":"ref10","article-title":"Vision-based manipulators need to also see from their hands","author":"Hsu","year":"2022"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3213246"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/MRA.2007.339604"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1142\/S0219843612500065"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1201\/9781315136370"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1038\/nature14539"},{"key":"ref16","volume-title":"Deep learning","author":"Goodfellow","year":"2016"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2017.10.013"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8460487"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2019.XV.074"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2022.xviii.010"},{"key":"ref21","first-page":"1992","article-title":"Visual imitation made easy","volume-title":"Conference on Robot Learning","author":"Young"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2023.xix.025"},{"key":"ref23","article-title":"Rt-2: Vision-language-action models transfer web knowledge to robotic control","author":"Brohan","year":"2023"},{"key":"ref24","article-title":"A generalist agent","author":"Reed","year":"2022"},{"key":"ref25","article-title":"Octo: An open-source generalist robot policy","author":"Ghosh","year":"2023"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1016\/j.aiopen.2022.10.001"},{"key":"ref28","first-page":"894","article-title":"Cliport: What and where pathways for robotic manipulation","volume-title":"Conference on Robot Learning","author":"Shridhar"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01337"},{"key":"ref30","first-page":"785","article-title":"Perceiver-actor: A multi-task transformer for robotic manipulation","volume-title":"Conference on Robot Learning","author":"Shridhar"},{"key":"ref31","article-title":"Spatial-language attention policies for efficient robot learning","author":"Parashar","year":"2023"},{"key":"ref32","article-title":"Gnfactor: Multi-task real robot learning with generalizable neural feature fields","author":"Ze","year":"2023"},{"key":"ref33","article-title":"3d diffuser actor: Policy diffusion with 3d scene representations","author":"Ke","year":"2024"},{"key":"ref34","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International conference on machine learning","author":"Radford"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-34372-9"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2024.XX.045"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/IROS45743.2020.9341370"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.3004787"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/IROS51168.2021.9636388"},{"key":"ref40","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International conference on machine learning","author":"Radford"},{"key":"ref41","article-title":"Roformer: Enhanced transformer with rotary position embedding","author":"Su","year":"2021"},{"key":"ref42","article-title":"Sequence level training with recurrent neural networks","author":"Ranzato","year":"2015"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2974707"},{"key":"ref44","article-title":"A modular robotic arm control stack for research: Franka-interface and frankapy","author":"Zhang","year":"2020"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2013.6696520"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1177\/0278364911406761"},{"key":"ref47","first-page":"991","article-title":"Bc-z: Zero-shot task generalization with robotic imitation learning","volume-title":"Conference on Robot Learning","author":"Jang"},{"key":"ref48","article-title":"Perceiver io: A general architecture for structured inputs & outputs","author":"Jaegle","year":"2021"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref50","article-title":"Large batch optimization for deep learning: Training bert in 76 minutes","author":"You","year":"2019"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.3390\/s21020413"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1145\/3503250"}],"event":{"name":"2024 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","location":"Abu Dhabi, United Arab Emirates","start":{"date-parts":[[2024,10,14]]},"end":{"date-parts":[[2024,10,18]]}},"container-title":["2024 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10801246\/10801290\/10802366.pdf?arnumber=10802366","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,26]],"date-time":"2024-12-26T07:00:45Z","timestamp":1735196445000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10802366\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,14]]},"references-count":52,"URL":"https:\/\/doi.org\/10.1109\/iros58592.2024.10802366","relation":{},"subject":[],"published":{"date-parts":[[2024,10,14]]}}}