{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T06:16:06Z","timestamp":1765520166057,"version":"3.48.0"},"reference-count":41,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iros60139.2025.11246626","type":"proceedings-article","created":{"date-parts":[[2025,11,27]],"date-time":"2025-11-27T18:54:45Z","timestamp":1764269685000},"page":"12224-12231","source":"Crossref","is-referenced-by-count":0,"title":["Ag2x2: Robust Agent-Agnostic Visual Representations for Zero-Shot Bimanual Manipulation"],"prefix":"10.1109","author":[{"given":"Ziyin","family":"Xiong","sequence":"first","affiliation":[{"name":"Beijing Institute for General Artificial Intelligence (BIGAI),National Key Laboratory of General Artificial Intelligence"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yinghan","family":"Chen","sequence":"additional","affiliation":[{"name":"Beijing Institute for General Artificial Intelligence (BIGAI),National Key Laboratory of General Artificial Intelligence"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Puhao","family":"Li","sequence":"additional","affiliation":[{"name":"Beijing Institute for General Artificial Intelligence (BIGAI),National Key Laboratory of General Artificial Intelligence"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yixin","family":"Zhu","sequence":"additional","affiliation":[{"name":"Peking University,School of Psychological and Cognitive Sciences"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tengyu","family":"Liu","sequence":"additional","affiliation":[{"name":"Beijing Institute for General Artificial Intelligence (BIGAI),National Key Laboratory of General Artificial Intelligence"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siyuan","family":"Huang","sequence":"additional","affiliation":[{"name":"Beijing Institute for General Artificial Intelligence (BIGAI),National Key Laboratory of General Artificial Intelligence"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"doi-asserted-by":"publisher","key":"ref1","DOI":"10.1109\/TPAMI.2023.3339515"},{"issue":"30","key":"ref2","first-page":"1","article-title":"A review of robot learning for manipulation: Challenges, representations, and algorithms","volume":"22","author":"Kroemer","year":"2021","journal-title":"Journal of Machine Learning Research (JMLR)"},{"doi-asserted-by":"publisher","key":"ref3","DOI":"10.1007\/s00521-021-06449-x"},{"year":"2025","author":"Li","article-title":"Controlvla: Few-shot object-centric adaptation for pre-trained vision-language-action models","key":"ref4"},{"doi-asserted-by":"publisher","key":"ref5","DOI":"10.15607\/rss.2023.xix.016"},{"volume-title":"Conference on Robot Learning (CoRL)","author":"Fu","article-title":"Mobile aloha: Learning bimanual mobile manipulation with low-cost whole-body teleoperation","key":"ref6"},{"volume-title":"Conference on Robot Learning (CoRL)","author":"Grannen","article-title":"Stabilize to act: Learning to coordinate for bimanual manipulation","key":"ref7"},{"doi-asserted-by":"publisher","key":"ref8","DOI":"10.1109\/ICRA57147.2024.10610763"},{"doi-asserted-by":"publisher","key":"ref9","DOI":"10.1109\/CVPR52734.2025.00656"},{"volume-title":"Proceedings of International Conference on Learning Representations (ICLR)","author":"Ma","article-title":"VIP: towards universal visual reward and representation via value-implicit pre-training","key":"ref10"},{"volume-title":"Conference on Robot Learning (CoRL)","author":"Nair","article-title":"R3M: A universal visual representation for robot manipulation","key":"ref11"},{"doi-asserted-by":"publisher","key":"ref12","DOI":"10.1109\/IROS58592.2024.10801835"},{"volume-title":"CoRL 2024 Workshop on Whole-Body Control and Bimanual Manipulation (CoRL 2024 WCBM)","author":"Grotz","article-title":"Peract2: Benchmarking and learning for robotic bimanual manipulation tasks","key":"ref13"},{"doi-asserted-by":"publisher","key":"ref14","DOI":"10.1016\/j.robot.2012.07.005"},{"doi-asserted-by":"publisher","key":"ref15","DOI":"10.15607\/RSS.2016.XII.019"},{"doi-asserted-by":"publisher","key":"ref16","DOI":"10.1109\/TMECH.2023.3263357"},{"volume-title":"Conference on Robot Learning (CoRL)","author":"Heidinger","article-title":"2handedafforder: Learning precise actionable bimanual affordances from human videos","key":"ref17"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.1109\/LRA.2023.3295991"},{"doi-asserted-by":"publisher","key":"ref19","DOI":"10.1109\/LRA.2024.3445630"},{"volume-title":"Proceedings of Advances in Neural Information Processing Systems (NeurIPS)","author":"Xie","article-title":"Deep imitation learning for bimanual robotic manipulation","key":"ref20"},{"doi-asserted-by":"publisher","key":"ref21","DOI":"10.15607\/RSS.2024.XX.078"},{"volume-title":"Proceedings of International Conference on Machine Learning (ICML)","author":"Laskin","article-title":"Curl: Contrastive unsupervised representations for reinforcement learning","key":"ref22"},{"volume-title":"Proceedings of International Conference on Machine Learning (ICML)","author":"Gelada","article-title":"Deepmdp: Learning continuous latent space models for representation learning","key":"ref23"},{"doi-asserted-by":"publisher","key":"ref24","DOI":"10.15607\/RSS.2022.XVIII.010"},{"doi-asserted-by":"publisher","key":"ref25","DOI":"10.1007\/s10514-015-9459-7"},{"doi-asserted-by":"publisher","key":"ref26","DOI":"10.1109\/ICRA40945.2020.9197331"},{"volume-title":"Proceedings of International Conference on Machine Learning (ICML)","author":"Shah","article-title":"Rrl: Resnet as representation for reinforcement learning","key":"ref27"},{"volume-title":"Proceedings of International Conference on Machine Learning (ICML)","author":"Parisi","article-title":"The unsurprising effectiveness of pre-trained vision models for control","key":"ref28"},{"volume-title":"Proceedings of International Conference on Machine Learning (ICML)","author":"Seo","article-title":"Reinforcement learning with action-free pre-training from videos","key":"ref29"},{"volume-title":"Proceedings of International Conference on Learning Representations (ICLR)","author":"Dosovitskiy","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","key":"ref30"},{"doi-asserted-by":"publisher","key":"ref31","DOI":"10.1007\/s11263-015-0816-y"},{"volume-title":"Proceedings of International Conference on Learning Representations (ICLR)","author":"Hu","article-title":"Lora: Low-rank adaptation of large language models","key":"ref32"},{"doi-asserted-by":"publisher","key":"ref33","DOI":"10.1109\/TPAMI.2020.2991965"},{"doi-asserted-by":"publisher","key":"ref34","DOI":"10.1109\/CVPR52733.2024.00938"},{"year":"2020","article-title":"YOLOv5","key":"ref35"},{"doi-asserted-by":"publisher","key":"ref36","DOI":"10.1109\/CVPR52729.2023.00289"},{"doi-asserted-by":"publisher","key":"ref37","DOI":"10.1109\/cvpr52688.2022.01704"},{"doi-asserted-by":"publisher","key":"ref38","DOI":"10.1109\/CVPR42600.2020.01146"},{"doi-asserted-by":"publisher","key":"ref39","DOI":"10.12794\/metadc1505267"},{"volume-title":"Proceedings of International Conference on Learning Representations (ICLR)","author":"Ma","article-title":"Eureka: Human-level reward design via coding large language models","key":"ref40"},{"doi-asserted-by":"publisher","key":"ref41","DOI":"10.1037\/11491-005"}],"event":{"name":"2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","start":{"date-parts":[[2025,10,19]]},"location":"Hangzhou, China","end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11245651\/11245652\/11246626.pdf?arnumber=11246626","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T06:12:30Z","timestamp":1765519950000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11246626\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":41,"URL":"https:\/\/doi.org\/10.1109\/iros60139.2025.11246626","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}