{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T07:59:15Z","timestamp":1767772755371,"version":"3.28.2"},"reference-count":71,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T00:00:00Z","timestamp":1715558400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T00:00:00Z","timestamp":1715558400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100009002","name":"Shanghai University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100009002","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,5,13]]},"DOI":"10.1109\/icra57147.2024.10609992","type":"proceedings-article","created":{"date-parts":[[2024,8,8]],"date-time":"2024-08-08T17:51:05Z","timestamp":1723139465000},"page":"4318-4325","source":"Crossref","is-referenced-by-count":8,"title":["Object-Centric Instruction Augmentation for Robotic Manipulation"],"prefix":"10.1109","author":[{"given":"Junjie","family":"Wen","sequence":"first","affiliation":[{"name":"East China Normal University,School of Computer Science,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yichen","family":"Zhu","sequence":"additional","affiliation":[{"name":"Midea Group,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Minjie","family":"Zhu","sequence":"additional","affiliation":[{"name":"East China Normal University,School of Computer Science,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jinming","family":"Li","sequence":"additional","affiliation":[{"name":"Shanghai University,School of Science,Department of Mathematics,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhiyuan","family":"Xu","sequence":"additional","affiliation":[{"name":"Midea Group,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhengping","family":"Che","sequence":"additional","affiliation":[{"name":"Midea Group,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chaomin","family":"Shen","sequence":"additional","affiliation":[{"name":"East China Normal University,School of Computer Science,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yaxin","family":"Peng","sequence":"additional","affiliation":[{"name":"Shanghai University,School of Science,Department of Mathematics,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dong","family":"Liu","sequence":"additional","affiliation":[{"name":"Midea Group,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feifei","family":"Feng","sequence":"additional","affiliation":[{"name":"Midea Group,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jian","family":"Tang","sequence":"additional","affiliation":[{"name":"Midea Group,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/0959-4388(94)90066-3"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.tics.2011.08.005"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1523\/JNEUROSCI.1462-12.2012"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/0166-2236(92)90344-8"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.3389\/fnint.2016.00037"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.7554\/eLife.34464"},{"key":"ref7","first-page":"20","article-title":"Chatgpt for robotics: Design principles and model abilities","volume":"2","author":"Vemprala","year":"2023","journal-title":"Microsoft Auton. Syst. Robot. Res"},{"article-title":"Do as i can, not as i say: Grounding language in robotic affordances","year":"2022","author":"Ahn","key":"ref8"},{"article-title":"Embodiedgpt: Vision-language pre-training via embodied chain of thought","year":"2023","author":"Mu","key":"ref9"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2023.xix.029"},{"key":"ref11","first-page":"24824","article-title":"Chain-of-thought prompting elicits reasoning in large language models","volume-title":"Advances in Neural Information Processing Systems","volume":"35","author":"Wei"},{"key":"ref12","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International conference on machine learning","author":"Radford"},{"key":"ref13","first-page":"22199","article-title":"Large language models are zero-shot reasoners","volume-title":"Advances in neural information processing systems","volume":"35","author":"Kojima"},{"key":"ref14","first-page":"18343","article-title":"Minedojo: Building open-ended embodied agents with internet-scale knowledge","volume-title":"Advances in Neural Information Processing Systems","volume":"35","author":"Fan"},{"key":"ref15","first-page":"894","article-title":"Cliport: What and where pathways for robotic manipulation","volume-title":"Conference on Robot Learning","author":"Shridhar"},{"key":"ref16","doi-asserted-by":"crossref","DOI":"10.1145\/3688863.3689575","article-title":"Llava-phi: Efficient multi-modal assistant with small language model","author":"Zhu","year":"2024"},{"article-title":"Query-relevant images jailbreak large multi-modal models","year":"2023","author":"Liu","key":"ref17"},{"key":"ref18","first-page":"9118","article-title":"Language models as zero-shot planners: Extracting actionable knowledge for embodied agents","volume-title":"International Conference on Machine Learning","author":"Huang"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/IROS55552.2023.10341472"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160969"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160591"},{"article-title":"Roco: Dialectic multi-robot collaboration with large language models","year":"2023","author":"Mandi","key":"ref22"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/3568162.3578623"},{"article-title":"Palm-e: An embodied multimodal language model","year":"2023","author":"Driess","key":"ref24"},{"article-title":"Vima: General robot manipulation with multimodal prompts","year":"2022","author":"Jiang","key":"ref25"},{"article-title":"Human instruction-following with deep reinforcement learning via transfer-learning from text","year":"2020","author":"Hill","key":"ref26"},{"key":"ref27","first-page":"991","article-title":"Bc-z: Zero-shot task generalization with robotic imitation learning","volume-title":"Conference on Robot Learning","author":"Jang"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1055\/a-1858-6495"},{"key":"ref29","first-page":"1303","article-title":"Learning language-conditioned robot behavior from offline data and crowd-sourced annotation","volume-title":"Conference on Robot Learning","author":"Nair"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2021.xvii.047"},{"key":"ref31","first-page":"785","article-title":"Perceiver-actor: A multitask transformer for robotic manipulation","volume-title":"Conference on Robot Learning","author":"Shridhar"},{"article-title":"Vip: Towards universal visual reward and representation via value-implicit pre-training","year":"2022","author":"Ma","key":"ref32"},{"key":"ref33","first-page":"19884","article-title":"Reinforcement learning with augmented data","volume-title":"Advances in neural information processing systems","volume":"33","author":"Laskin"},{"article-title":"Rrl: Resnet as representation for reinforcement learning","year":"2021","author":"Shah","key":"ref34"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2023.XIX.032"},{"issue":"4","key":"ref36","first-page":"7","article-title":"Clip on wheels: Zero-shot object navigation as object localization and exploration","volume":"3","author":"Gadre","year":"2022"},{"article-title":"Open-world object manipulation using pre-trained vision-language models","year":"2023","author":"Stone","key":"ref37"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161534"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10611525"},{"article-title":"Vision-language models as success detectors","year":"2023","author":"Du","key":"ref40"},{"article-title":"Grounding classical task planners via vision-language models","year":"2023","author":"Zhang","key":"ref41"},{"article-title":"Distilling internet-scale vision-language models into embodied agents","year":"2023","author":"Sumers","key":"ref42"},{"article-title":"Rt-2: Visionlanguage-action models transfer web knowledge to robotic control","year":"2023","author":"Brohan","key":"ref43"},{"article-title":"Deep object pose estimation for semantic robotic grasping of household objects","year":"2018","author":"Tremblay","key":"ref44"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/IROS47612.2022.9981838"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2965875"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8794224"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"article-title":"Learning generalizable manipulation policies with object-centric 3d representations","volume-title":"7th Annual Conference on Robot Learning","author":"Zhu","key":"ref49"},{"author":"Shi","key":"ref50","article-title":"Plug-and-play object-centric representations from \u201cwhat\u201d and \u201cwhere\u201d foundation models"},{"key":"ref51","first-page":"11525","article-title":"Objectcentric learning with slot attention","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Locatello"},{"article-title":"Monet: Unsupervised scene decomposition and representation","year":"2019","author":"Burgess","key":"ref52"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/IROS51168.2021.9636023"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160888"},{"article-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","year":"2023","author":"Zhu","key":"ref55"},{"article-title":"Visual instruction tuning","year":"2023","author":"Liu","key":"ref56"},{"key":"ref57","article-title":"Im2text: Describing images using 1 million captioned photographs","volume-title":"Advances in neural information processing systems","volume":"24","author":"Ordonez"},{"key":"ref58","first-page":"25278","article-title":"Laion-5b: An open large-scale dataset for training next generation image-text models","volume-title":"Advances in Neural Information Processing Systems","volume":"35","author":"Schuhmann"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_5"},{"key":"ref60","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","volume-title":"Advances in Neural Information Processing Systems","volume":"35","author":"Ouyang"},{"article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","year":"2020","author":"Dosovitskiy","key":"ref61"},{"article-title":"Llama: Open and efficient foundation language models","year":"2023","author":"Touvron","key":"ref62"},{"article-title":"Llama 2: Open foundation and fine-tuned chat models","year":"2023","author":"Touvron","key":"ref63"},{"article-title":"Layer normalization","year":"2016","author":"Ba","key":"ref64"},{"article-title":"Disclip: Open-vocabulary referring expression generation","year":"2023","author":"Bracha","key":"ref65"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_40"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20240"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01335"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548096"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58595-2_32"},{"article-title":"R3m: A universal visual representation for robot manipulation","year":"2022","author":"Nair","key":"ref71"}],"event":{"name":"2024 IEEE International Conference on Robotics and Automation (ICRA)","start":{"date-parts":[[2024,5,13]]},"location":"Yokohama, Japan","end":{"date-parts":[[2024,5,17]]}},"container-title":["2024 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10609961\/10609862\/10609992.pdf?arnumber=10609992","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,26]],"date-time":"2024-11-26T13:29:52Z","timestamp":1732627792000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10609992\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,13]]},"references-count":71,"URL":"https:\/\/doi.org\/10.1109\/icra57147.2024.10609992","relation":{},"subject":[],"published":{"date-parts":[[2024,5,13]]}}}