{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:40:33Z","timestamp":1777657233915,"version":"3.51.4"},"reference-count":47,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,5,29]],"date-time":"2023-05-29T00:00:00Z","timestamp":1685318400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,5,29]],"date-time":"2023-05-29T00:00:00Z","timestamp":1685318400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,5,29]]},"DOI":"10.1109\/icra48891.2023.10160739","type":"proceedings-article","created":{"date-parts":[[2023,7,4]],"date-time":"2023-07-04T13:20:56Z","timestamp":1688476856000},"page":"3867-3874","source":"Crossref","is-referenced-by-count":11,"title":["Efficient Bimanual Handover and Rearrangement via Symmetry-Aware Actor-Critic Learning"],"prefix":"10.1109","author":[{"given":"Yunfei","family":"Li","sequence":"first","affiliation":[{"name":"Institute of Inter-disciplinary Information Sciences, Tsinghua University,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chaoyi","family":"Pan","sequence":"additional","affiliation":[{"name":"Tsinghua University,Department of Electronic Engineering,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huazhe","family":"Xu","sequence":"additional","affiliation":[{"name":"Institute of Inter-disciplinary Information Sciences, Tsinghua University,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaolong","family":"Wang","sequence":"additional","affiliation":[{"name":"UC,Department of Electrical and Computer Engineering,San Diego,CA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Wu","sequence":"additional","affiliation":[{"name":"Institute of Inter-disciplinary Information Sciences, Tsinghua University,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2016.7487583"},{"key":"ref35","article-title":"Toward compositional generalization in object-oriented world modeling","author":"zhao","year":"2022","journal-title":"ArXiv Preprint"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/HUMANOIDS.2014.7041499"},{"key":"ref34","first-page":"4199","article-title":"Mdp homomorphic networks: Group symmetries in reinforcement learning","volume":"33","author":"van der pol","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/70.704220"},{"key":"ref37","first-page":"2085","article-title":"Value-decomposition networks for cooperative multi-agent learning based on team reward","author":"sunehag","year":"2018","journal-title":"Proceedings of the 17th International Conference on Autonomous Agents and MultiAgent Systems AAMAS 2018 Stockholm Sweden July 10&#x2013;15 2018"},{"key":"ref14","first-page":"183","article-title":"Efficient obstacle rear-rangement for object manipulation tasks in cluttered environments","author":"lee","year":"0","journal-title":"2019 International Conference on Robotics and Automation (ICRA)"},{"key":"ref36","article-title":"Symmetry learning for function approximation in reinforcement learning","author":"mahajan","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref31","author":"ravindran","year":"2001","journal-title":"Symmetries and model minimization in Markov decision processes"},{"key":"ref30","first-page":"106","article-title":"Model minimization in markov decision processes","author":"dean","year":"1997","journal-title":"AAAI\/IAAI"},{"key":"ref11","article-title":"Rearrangement: A challenge for embodied ai","author":"batra","year":"2020","journal-title":"ArXiv Preprint"},{"key":"ref33","article-title":"Value iter-ation networks","volume":"29","author":"tamar","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref10","article-title":"Isaac gym: High performance gpu-based physics simulation for robot learning","author":"makoviychuk","year":"2021","journal-title":"ArXiv Preprint"},{"key":"ref32","author":"ravindran","year":"2004","journal-title":"An Algebraic Approach to Abstraction in Reinforcement Learning"},{"key":"ref2","first-page":"345","article-title":"Two arms are better than one: A behavior based control system for assistive bimanual manipulation","author":"edsinger","year":"2007","journal-title":"Recent Progress in Robotics Viable Robotic Service to Human"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.aaw0955"},{"key":"ref17","first-page":"211","article-title":"Large-scale multi-object re-arrangement","author":"huang","year":"0","journal-title":"2019 International Conference on Robotics and Automation (ICRA)"},{"key":"ref39","first-page":"1094","article-title":"Learning to achieve goals","author":"kaelbling","year":"1993","journal-title":"Proceedings of the 13th International Joint Conference on Artificial Intelligence"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2011.6094737"},{"key":"ref38","first-page":"4292","article-title":"QMIX: monotonic value function factorisation for deep multi-agent reinforcement learning","volume":"80","author":"rashid","year":"0","journal-title":"Proceedings of the 35th International Conference on Machine Learning ICML 2018 Stockholmsm &#x00E4;ssan Stockholm Sweden July 10&#x2013;15 2018 ser Proceedings of Machine Learning Research"},{"key":"ref19","article-title":"Visual foresight: Model-based deep reinforcement learning for vision-based robotic control","author":"ebert","year":"2018","journal-title":"ArXiv Preprint"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561516"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-66723-8_15"},{"key":"ref46","article-title":"Attention is all you need","volume":"30","author":"vaswani","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2976329"},{"key":"ref45","first-page":"7754","article-title":"Generalized hindsight for reinforcement learning","volume":"33","author":"li","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref26","first-page":"1334","article-title":"End-to-end training of deep visuomotor policies","volume":"17","author":"levine","year":"2016","journal-title":"The Journal of Machine Learning Research"},{"key":"ref25","article-title":"Long-horizon multi-robot rearrangement planning for construction assembly","author":"hartmann","year":"2021","journal-title":"ArXiv Preprint"},{"key":"ref47","doi-asserted-by":"crossref","first-page":"2280","DOI":"10.1016\/j.patcog.2014.01.005","article-title":"Automatic generation and detection of highly reliable fiducial markers under occlusion","volume":"47","author":"garrido-jurado","year":"2014","journal-title":"Pattern Recognition"},{"key":"ref20","first-page":"270","article-title":"Rear-rangement with nonprehensile manipulation using deep reinforcement learning","author":"yuan","year":"0","journal-title":"2018 IEEE International Conference on Robotics and Automation (ICRA)"},{"key":"ref42","article-title":"Mher: Model-based hindsight experience replay","author":"yang","year":"0","journal-title":"Deep RL Workshop NeurIPS 2021"},{"key":"ref41","article-title":"Temporal difference models: Model-free deep RL for model-based control","author":"pong","year":"0","journal-title":"6th International conference on Learning Representations ICLR 2018 Vancouver BC Canada April 30 - May 3 2018 Conference Track Proceedings"},{"key":"ref22","article-title":"Asymmetric self-play for automatic goal discovery in robotic manipulation","author":"openai","year":"2021","journal-title":"ArXiv Preprint"},{"key":"ref44","first-page":"14783","article-title":"Rewriting history with inverse rl: Hindsight inference for policy improvement","volume":"33","author":"eysenbach","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref21","first-page":"726","article-title":"Transporter networks: Rearranging the visual world for robotic manipulation","author":"zeng","year":"0","journal-title":"Conference on Robot Learning"},{"key":"ref43","article-title":"Curriculum-guided hindsight experience replay","volume":"32","author":"fang","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2020.XVI.065"},{"key":"ref27","article-title":"Bi-manual manipulation and attachment via sim-to-real reinforcement learning","author":"kataoka","year":"2022","journal-title":"ArXiv Preprint"},{"key":"ref29","article-title":"Dair: Disentangled attention intrinsic regularization for safe and efficient bimanual manipulation","author":"zhang","year":"2021","journal-title":"ArXiv Preprint"},{"key":"ref8","article-title":"Hind-sight experience replay","volume":"30","author":"andrychowicz","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561491"},{"key":"ref9","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","author":"haarnoja","year":"0","journal-title":"International Conference on Machine Learning"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196958"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2012.07.005"},{"key":"ref6","article-title":"Deep imitation learning for bimanual robotic manipulation","author":"xie","year":"0","journal-title":"Advances in Neural Information Processing Systems 33 Annual Conference on Neural Information Processing Systems 2020 NeurIPS 2020 December 6&#x2013;12 2020 virtual"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-26326-3"},{"key":"ref40","first-page":"1312","article-title":"Universal value function approximators","volume":"37","author":"schaul","year":"0","journal-title":"Proceedings of the 32nd International Conference on Machine Learning ICML 2015 Lille France 6&#x2013;11 July 2015 ser JMLR Workshop and Conference Proceedings"}],"event":{"name":"2023 IEEE International Conference on Robotics and Automation (ICRA)","location":"London, United Kingdom","start":{"date-parts":[[2023,5,29]]},"end":{"date-parts":[[2023,6,2]]}},"container-title":["2023 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10160211\/10160212\/10160739.pdf?arnumber=10160739","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,24]],"date-time":"2023-07-24T13:31:32Z","timestamp":1690205492000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10160739\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,5,29]]},"references-count":47,"URL":"https:\/\/doi.org\/10.1109\/icra48891.2023.10160739","relation":{},"subject":[],"published":{"date-parts":[[2023,5,29]]}}}