{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T05:13:44Z","timestamp":1780377224570,"version":"3.54.1"},"reference-count":45,"publisher":"Informa UK Limited","issue":"8","content-domain":{"domain":["www.tandfonline.com"],"crossmark-restriction":true},"short-container-title":["Advanced Robotics"],"published-print":{"date-parts":[[2022,4,18]]},"DOI":"10.1080\/01691864.2022.2046504","type":"journal-article","created":{"date-parts":[[2022,3,10]],"date-time":"2022-03-10T19:26:20Z","timestamp":1646940380000},"page":"404-421","update-policy":"https:\/\/doi.org\/10.1080\/tandf_crossmark_01","source":"Crossref","is-referenced-by-count":3,"title":["Residual reinforcement learning for logistics cart transportation"],"prefix":"10.1080","volume":"36","author":[{"given":"Ryosuke","family":"Matsuo","sequence":"first","affiliation":[{"name":"Department of Aeronautics and Astronautics, The University of Tokyo, Tokyo, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shinya","family":"Yasuda","sequence":"additional","affiliation":[{"name":"System Platform Research Laboratories, NEC Corporation, Kanagawa, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Taichi","family":"Kumagai","sequence":"additional","affiliation":[{"name":"System Platform Research Laboratories, NEC Corporation, Kanagawa, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Natsuhiko","family":"Sato","sequence":"additional","affiliation":[{"name":"System Platform Research Laboratories, NEC Corporation, Kanagawa, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hiroshi","family":"Yoshida","sequence":"additional","affiliation":[{"name":"System Platform Research Laboratories, NEC Corporation, Kanagawa, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Takehisa","family":"Yairi","sequence":"additional","affiliation":[{"name":"Department of Aeronautics and Astronautics, The University of Tokyo, Tokyo, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"301","published-online":{"date-parts":[[2022,3,10]]},"reference":[{"key":"CIT0001","doi-asserted-by":"publisher","DOI":"10.1109\/IECON.2019.8927467"},{"key":"CIT0002","unstructured":"Lillicrap TP, Hunt JJ, Pritzel A, et\u00a0al. Continuous control with deep reinforcement learning. Preprint 2015. Available from: arXiv:1509.02971."},{"key":"CIT0003","unstructured":"Schulman J, Levine S, Abbeel P, et\u00a0al. Trust region policy optimization. International conference on machine learning; PMLR; 2015. p.\u00a01889\u20131897."},{"key":"CIT0004","unstructured":"Mnih V, Puigdomenech Badia A, Mirza M, et\u00a0al. Asynchronous methods for deep reinforcement learning. International conference on machine learning; PMLR; 2016. p.\u00a01928\u20131937."},{"key":"CIT0005","unstructured":"Schulman J, Wolski F, Dhariwal P, et\u00a0al. Proximal policy optimization algorithms. Preprint 2017. Available from: arXiv:1707.06347."},{"key":"CIT0006","unstructured":"Fujimoto S, Hoof H, Meger D. Addressing function approximation error in actor-critic methods. International Conference on Machine Learning; PMLR; 2018. p.\u00a01587\u20131596."},{"key":"CIT0007","unstructured":"Haarnoja T, Zhou A, Abbeel P, et\u00a0al. Soft actor-critic: off-policy maximum entropy deep reinforcement learning with a stochastic actor. International Conference on Machine Learning; PMLR; 2018. p.\u00a01861\u20131870."},{"key":"CIT0008","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"CIT0009","unstructured":"Coumans E, Bai Y. Pybullet, a python module for physics simulation for games, robotics and machine learning. 2016\u20132021. Available from: http:\/\/pybullet.org."},{"key":"CIT0010","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2013.6696520"},{"key":"CIT0011","unstructured":"Koenig N, Howard A. Design and use paradigms for gazebo, an open-source multi-robot simulator. 2004 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)(IEEE Cat. No.\u00a004CH37566); Vol.\u00a03, IEEE; 2004. p.\u00a02149\u20132154."},{"key":"CIT0012","unstructured":"Juliani A, Berges V-P, Teng E, et\u00a0al. Unity: a general platform for intelligent agents. preprint 2018. Available from: arXiv:1809.02627."},{"key":"CIT0013","unstructured":"Nair A, Srinivasan P, Blackwell S, et\u00a0al. Massively parallel methods for deep reinforcement learning. International Conference on Machine Learning Deep Learning Workshop; 2015."},{"key":"CIT0014","unstructured":"Horgan D, Quan J, Budden D, et\u00a0al. Distributed prioritized experience replay. International Conference on Learning Representations (ICLR); 2018."},{"key":"CIT0015","unstructured":"Espeholt L, Soyer H, Munos R, et\u00a0al. Impala: scalable distributed deep-rl with importance weighted actor-learner architectures. International Conference on Machine Learning; PMLR; 2018. p.\u00a01407\u20131416."},{"key":"CIT0016","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8794127"},{"key":"CIT0017","unstructured":"Silver T, Allen K, Tenenbaum J, et\u00a0al. Residual policy learning. Preprint 2018. Available from: arXiv:1812.06298."},{"key":"CIT0018","doi-asserted-by":"publisher","DOI":"10.1109\/TRO.2020.2988642"},{"key":"CIT0019","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8793864"},{"key":"CIT0020","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2020.2977374"},{"key":"CIT0021","unstructured":"Lowe R, Wu Y, Tamar A, et\u00a0al. Multi-agent actor-critic for mixed cooperative-competitive environments. Preprint 2017. Available from: arXiv:1706.02275."},{"key":"CIT0022","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-71682-4_5"},{"key":"CIT0023","doi-asserted-by":"publisher","DOI":"10.3389\/frobt.2018.00059"},{"key":"CIT0024","doi-asserted-by":"publisher","DOI":"10.1109\/ROBIO.2014.7090309"},{"key":"CIT0025","doi-asserted-by":"crossref","unstructured":"Kosuge K, Oosumi T. Decentralized control of multiple robots handling an object. Proceedings of IEEE\/RSJ International Conference on Intelligent Robots and Systems. IROS'96; Vol.\u00a01, IEEE; 1996. p.\u00a0318\u2013323.","DOI":"10.1109\/IROS.1996.570694"},{"key":"CIT0026","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2005.1545269"},{"key":"CIT0027","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2016.7487477"},{"key":"CIT0028","doi-asserted-by":"publisher","DOI":"10.5772\/60972"},{"key":"CIT0029","doi-asserted-by":"crossref","unstructured":"Brown RG, Jennings JS. A pusher\/steerer model for strongly cooperative mobile robot manipulation. Proceedings 1995 IEEE\/RSJ International Conference on Intelligent Robots and Systems; Human Robot Interaction and Cooperative Robots. Vol.\u00a03, IEEE; 1995. p.\u00a0562\u2013568.","DOI":"10.1109\/IROS.1995.525941"},{"key":"CIT0030","doi-asserted-by":"crossref","unstructured":"Wang Z, Hirata Y, Kosuge K. Control multiple mobile robots for object caging and manipulation. Proceedings 2003 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS 2003)(Cat. No.\u00a003CH37453); Vol.\u00a02, IEEE; 2003. p.\u00a01751\u20131756.","DOI":"10.1109\/IROS.2003.1248897"},{"key":"CIT0031","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2017.2733552"},{"key":"CIT0032","doi-asserted-by":"publisher","DOI":"10.1080\/01691864.2019.1633955"},{"key":"CIT0033","doi-asserted-by":"publisher","DOI":"10.1177\/02783649922066222"},{"key":"CIT0034","doi-asserted-by":"crossref","unstructured":"Pipattanasomporn P, Sudsang A. Two-finger caging of concave polygon. Proceedings 2006 IEEE International Conference on Robotics and Automation, 2006. ICRA 2006; IEEE; 2006. p.\u00a02137\u20132142.","DOI":"10.1109\/ROBOT.2006.1642020"},{"key":"CIT0035","doi-asserted-by":"publisher","DOI":"10.1109\/TRO.2015.2463651"},{"key":"CIT0036","doi-asserted-by":"publisher","DOI":"10.1080\/01691864.2017.1371075"},{"key":"CIT0037","doi-asserted-by":"crossref","DOI":"10.1109\/ACCESS.2020.3025287","volume":"8","author":"Zhang L","year":"2020","journal-title":"IEEE Access"},{"key":"CIT0038","doi-asserted-by":"publisher","DOI":"10.1016\/j.oceaneng.2019.04.099"},{"key":"CIT0039","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2967299"},{"key":"CIT0040","unstructured":"Kanayama Y, Kimura Y, Miyazaki F, et\u00a0al. A stable tracking control method for an autonomous mobile robot. Proceedings., IEEE International Conference on Robotics and Automation; IEEE; 1990. p.\u00a0384\u2013389."},{"key":"CIT0041","unstructured":"Bergstra J, Bardenet R, Bengio Y, et\u00a0al. Algorithms for hyper-parameter optimization. 25th annual conference on neural information processing systems (NIPS 2011); Vol.\u00a024, Neural Information Processing Systems Foundation; 2011."},{"key":"CIT0042","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330701"},{"key":"CIT0043","unstructured":"Silver D, Lever G, Heess N, et\u00a0al. Deterministic policy gradient algorithms. International conference on machine learning; PMLR; 2014. p.\u00a0387\u2013395."},{"key":"CIT0044","unstructured":"Schaul T, Quan J, Antonoglou I, et\u00a0al. Prioritized experience replay. International Conference on Learning Representations (ICLR); 2016."},{"key":"CIT0045","unstructured":"Kingma DP, Ba J. Adam: a method for stochastic optimization. ICLR (Poster). 2015."}],"container-title":["Advanced Robotics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.tandfonline.com\/doi\/pdf\/10.1080\/01691864.2022.2046504","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,20]],"date-time":"2024-09-20T02:09:59Z","timestamp":1726798199000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.tandfonline.com\/doi\/full\/10.1080\/01691864.2022.2046504"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,3,10]]},"references-count":45,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2022,4,18]]}},"alternative-id":["10.1080\/01691864.2022.2046504"],"URL":"https:\/\/doi.org\/10.1080\/01691864.2022.2046504","relation":{},"ISSN":["0169-1864","1568-5535"],"issn-type":[{"value":"0169-1864","type":"print"},{"value":"1568-5535","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,3,10]]},"assertion":[{"value":"The publishing and review policy for this title is described in its Aims & Scope.","order":1,"name":"peerreview_statement","label":"Peer Review Statement"},{"value":"http:\/\/www.tandfonline.com\/action\/journalInformation?show=aimsScope&journalCode=tadr20","URL":"http:\/\/www.tandfonline.com\/action\/journalInformation?show=aimsScope&journalCode=tadr20","order":2,"name":"aims_and_scope_url","label":"Aim & Scope"},{"value":"2021-06-13","order":0,"name":"received","label":"Received","group":{"name":"publication_history","label":"Publication History"}},{"value":"2021-11-08","order":1,"name":"revised","label":"Revised","group":{"name":"publication_history","label":"Publication History"}},{"value":"2022-02-04","order":2,"name":"accepted","label":"Accepted","group":{"name":"publication_history","label":"Publication History"}},{"value":"2022-03-10","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}