{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:09:23Z","timestamp":1784268563958,"version":"3.55.0"},"reference-count":87,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T00:00:00Z","timestamp":1715558400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T00:00:00Z","timestamp":1715558400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100017090","name":"Sony","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100017090","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,5,13]]},"DOI":"10.1109\/icra57147.2024.10610566","type":"proceedings-article","created":{"date-parts":[[2024,8,8]],"date-time":"2024-08-08T17:51:05Z","timestamp":1723139465000},"page":"6672-6679","source":"Crossref","is-referenced-by-count":18,"title":["Gen2Sim: Scaling up Robot Learning in Simulation with Generative Models"],"prefix":"10.1109","author":[{"given":"Pushkal","family":"Katara","sequence":"first","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhou","family":"Xian","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Katerina","family":"Fragkiadaki","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Language models are few-shot learners","author":"Brown","year":"2020"},{"key":"ref2","article-title":"Learning transferable visual models from natural language supervision","volume-title":"CoRR","volume":"abs\/2103.00020","author":"Radford","year":"2021"},{"key":"ref3","article-title":"Robust speech recognition via large-scale weak supervision","author":"Radford","year":"2022"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1177\/0278364917710318"},{"key":"ref6","first-page":"979","article-title":"Graph-structured visual imitation","volume-title":"Conference on Robot Learning","author":"Sieb"},{"key":"ref7","author":"Brohan","year":"2023","journal-title":"Rt-2: Vision-language-action models transfer web knowledge to robotic control"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2023.xix.025"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/tase.2021.3064065"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01676"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2021.XVII.011"},{"key":"ref12","article-title":"Learning quadrupedal locomotion over challenging terrain","volume-title":"CoRR","volume":"abs\/2010.11251","author":"Lee","year":"2020"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160325"},{"key":"ref14","first-page":"297","article-title":"A system for general in-hand object re-orientation","volume-title":"Conference on Robot Learning","author":"Chen"},{"key":"ref15","article-title":"Solving rubik\u2019s cube with a robot hand","author":"Akkaya","year":"2019"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-023-06419-4"},{"key":"ref17","article-title":"Carla: An open urban driving simulator","author":"Dosovitskiy","year":"2017"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01111"},{"key":"ref19","article-title":"Behavior: Benchmark for everyday household activities in virtual, interactive, and ecological environments","author":"Srivastava","year":"2021"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00943"},{"key":"ref21","article-title":"Threedworld: A platform for interactive multi-modal physical simulation","author":"Gan","year":"2021"},{"key":"ref22","author":"Xian","year":"2023","journal-title":"Towards a foundation model for generalist robots: Diverse skill learning at scale via automated task and scene generation"},{"key":"ref23","article-title":"Hierarchical planning for long-horizon manipulation with geometric and symbolic scene graphs","volume-title":"CoRR","volume":"abs\/2012.07277","author":"Zhu","year":"2020"},{"key":"ref24","article-title":"Regression planning networks","volume-title":"CoRR","volume":"abs\/1909.13072","author":"Xu","year":"2019"},{"key":"ref25","author":"Huang","year":"2022","journal-title":"Language models as zero-shot planners: Extracting actionable knowledge for embodied agents"},{"key":"ref26","article-title":"Inner monologue: Embodied reasoning through planning with language models","author":"Huang","year":"2022"},{"key":"ref27","article-title":"Code as policies: Language model programs for embodied control","author":"Liang","year":"2022"},{"key":"ref28","article-title":"Chain of thought prompting elicits reasoning in large language models","volume-title":"CoRR","volume":"abs\/2201.11903","author":"Wei","year":"2022"},{"key":"ref29","article-title":"Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing","volume-title":"CoRR","volume":"abs\/2107.13586","author":"Liu","year":"2021"},{"key":"ref30","article-title":"Language models are few-shot learners","author":"Brown","year":"2020"},{"key":"ref31","article-title":"Data distributional properties drive emergent in-context learning in transformers","author":"Chan","year":"2022"},{"key":"ref32","article-title":"Ghost in the minecraft: Generally capable agents for open-world environments via large language models with text-based knowledge and memory","author":"Zhu","year":"2023"},{"key":"ref33","article-title":"Steve-1: A generative model for text-to-behavior in minecraft","author":"Lifshitz","year":"2023"},{"key":"ref34","article-title":"Voyager: An open-ended embodied agent with large language models","author":"Wang","year":"2023"},{"key":"ref35","author":"Ha","year":"2023","journal-title":"Scaling up and distilling down: Language-guided robot skill acquisition"},{"key":"ref36","author":"Huang","year":"2023","journal-title":"Voxposer: Composable 3d value maps for robotic manipulation with language models"},{"key":"ref37","author":"Yu","year":"2023","journal-title":"Language to rewards for robotic skill synthesis"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00768"},{"key":"ref39","first-page":"5799","article-title":"pigan: Periodic implicit generative adversarial networks for 3d-aware image synthesis","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","author":"Chan"},{"key":"ref40","author":"Poole","year":"2022","journal-title":"Dreamfusion: Text-to-3d using 2d diffusion"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02033"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00037"},{"key":"ref43","doi-asserted-by":"crossref","DOI":"10.1109\/CVPR52729.2023.00816","article-title":"Realfusion: 360 reconstruction of any object from a single image","volume-title":"CVPR","author":"Melas-Kyriazi","year":"2023"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02086"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00476"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9812293"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1093\/oed\/6707140721"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/IROS55552.2023.10341811"},{"key":"ref49","article-title":"Motion policy networks","author":"Fishman","year":"2022"},{"key":"ref50","article-title":"Imitating task and motion planning with visuomotor transformers","author":"Dalal","year":"2023"},{"key":"ref51","article-title":"Guided imitation of task and motion planning","author":"McDonald","year":"2021"},{"key":"ref52","article-title":"Object-centric task and motion planning in dynamic environments","volume-title":"CoRR","volume":"abs\/1911.04679","author":"Migimatsu","year":"2019"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2011.5980391"},{"key":"ref54","first-page":"1930","article-title":"Logic-geometric programming: An optimization-based approach to combined task and motion planning","volume-title":"Proceedings of the 24th International Conference on Artificial Intelligence","author":"Toussaint"},{"key":"ref55","article-title":"SDRL: interpretable and data-efficient deep reinforcement learning leveraging symbolic planning","volume-title":"CoRR","volume":"abs\/1811.00090","author":"Lyu","year":"2018"},{"key":"ref56","article-title":"Stripstream: Integrating symbolic planners and blackbox samplers","volume-title":"CoRR","volume":"abs\/1802.08705","author":"Garrett","year":"2018"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161528"},{"key":"ref58","author":"Brockman","year":"2016","journal-title":"Openai gym"},{"key":"ref59","author":"Kolve","year":"2017","journal-title":"Ai2-thor: An interactive 3d environment for visual ai"},{"key":"ref60","article-title":"Pybullet, a python module for physics simulation for games, robotics and machine learning","author":"Coumans","year":"2016"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref62","author":"Tung","year":"2020","journal-title":"3d-oes: Viewpoint-invariant object-factorized environment simulators"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2023.XIX.030"},{"key":"ref64","author":"Xian","year":"2021","journal-title":"Hyperdynamics: Meta-learning object and agent dynamics with hypernetworks"},{"key":"ref65","author":"Gervet","year":"2023","journal-title":"Act3d: Infinite resolution action detection transformer for robotic manipulation"},{"key":"ref66","article-title":"Chaineddiffuser: Unifying trajectory diffusion and keypose prediction for robotic manipulation","volume-title":"Conference on Robot Learning","author":"Xian"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1145\/2601097.2601152"},{"key":"ref68","author":"Gan","year":"2020","journal-title":"Threedworld: A platform for interactive multi-modal physical simulation"},{"key":"ref69","author":"Lin","year":"2020","journal-title":"Softgym: Benchmarking deep reinforcement learning for deformable object manipulation"},{"key":"ref70","article-title":"Fluidlab: A differentiable environment for benchmarking complex fluid manipulation","volume-title":"International Conference on Learning Representations","author":"Xian"},{"key":"ref71","article-title":"Softzoo: A soft robot co-design benchmark for locomotion in diverse environments","volume-title":"International Conference on Learning Representations","author":"Wang"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref73","article-title":"Gpt-4 technical report","year":"2023"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00853"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01214"},{"issue":"4","key":"ref78","first-page":"1","article-title":"Instant neural graphics primitives with a multiresolution hash encoding","volume-title":"ACM Transactions on Graphics","volume":"41","author":"M\u00fcller","year":"2022"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1145\/3588432.3591503"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2017.8202133"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.2312\/LocalChapterEvents\/ItalChap\/ItalianChapConf2008\/129-136"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00684"},{"key":"ref83","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00931"},{"key":"ref85","author":"Jaegle","year":"2021","journal-title":"Perceiver io: A general architecture for structured inputs & outputs"},{"key":"ref86","article-title":"An image is worth one word: Per-sonalizing text-to-image generation using textual inversion","author":"Gal","year":"2022"},{"key":"ref87","article-title":"Isaac gym: High performance gpu-based physics simulation for robot learning","author":"Makoviychuk","year":"2021"}],"event":{"name":"2024 IEEE International Conference on Robotics and Automation (ICRA)","location":"Yokohama, Japan","start":{"date-parts":[[2024,5,13]]},"end":{"date-parts":[[2024,5,17]]}},"container-title":["2024 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10609961\/10609862\/10610566.pdf?arnumber=10610566","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,10]],"date-time":"2024-08-10T05:21:01Z","timestamp":1723267261000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10610566\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,13]]},"references-count":87,"URL":"https:\/\/doi.org\/10.1109\/icra57147.2024.10610566","relation":{},"subject":[],"published":{"date-parts":[[2024,5,13]]}}}