{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T14:34:13Z","timestamp":1785767653104,"version":"3.56.0"},"reference-count":20,"publisher":"IEEE","license":[{"start":{"date-parts":[[2018,8,1]],"date-time":"2018-08-01T00:00:00Z","timestamp":1533081600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2018,8,1]],"date-time":"2018-08-01T00:00:00Z","timestamp":1533081600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018,8]]},"DOI":"10.1109\/rcar.2018.8621776","type":"proceedings-article","created":{"date-parts":[[2019,1,25]],"date-time":"2019-01-25T02:29:36Z","timestamp":1548383376000},"page":"34-41","source":"Crossref","is-referenced-by-count":17,"title":["Towards High Level Skill Learning: Learn to Return Table Tennis Ball Using Monte-Carlo Based Policy Gradient Method"],"prefix":"10.1109","author":[{"given":"Yifeng","family":"Zhu","sequence":"first","affiliation":[{"name":"Zhejiang University, State Key Laboratory of Industrial Control and Technology, Hangzhou, 310027, P. R. China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongsheng","family":"Zhao","sequence":"additional","affiliation":[{"name":"Binhai Industrial Technology Research Institute of Zhejiang University, Tian, 300457, P. R. China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lisen","family":"Jin","sequence":"additional","affiliation":[{"name":"Zhejiang University, State Key Laboratory of Industrial Control and Technology, Hangzhou, 310027, P. R. China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jun","family":"Wu","sequence":"additional","affiliation":[{"name":"Zhejiang University, State Key Laboratory of Industrial Control and Technology, Hangzhou, 310027, P. R. China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rong","family":"Xiong","sequence":"additional","affiliation":[{"name":"Zhejiang University, State Key Laboratory of Industrial Control and Technology, Hangzhou, 310027, P. R. China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2007.11.026"},{"key":"ref11","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","author":"mnih","year":"2016","journal-title":"International Conference on Machine Learning"},{"key":"ref12","first-page":"387","article-title":"Deterministic policy gradient algorithms","author":"silver","year":"2014","journal-title":"Proceedings of the 31st International Conference on Machine Learning (ICML-14)"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.2174\/1573399812666160613113556"},{"key":"ref14","author":"gu","year":"2016","journal-title":"Deep reinforcement learning for robotic manipulation with asynchronous off-policy updates"},{"key":"ref15","first-page":"278","article-title":"Policy invariance under reward transformations: Theory and application to reward shaping","volume":"99","author":"ng","year":"1999","journal-title":"ICML"},{"key":"ref16","first-page":"2650","article-title":"Reinforcement learning to adjust robot movements to new situations","volume":"22","author":"kober","year":"2011","journal-title":"IJCAI Proceedings-International Joint Conference on Artificial Intelligence"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICHR.2010.5686298"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TIM.2014.2386951"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TIM.2016.2555179"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1177\/0278364912472380"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1177\/1059712311419378"},{"key":"ref6","doi-asserted-by":"crossref","first-page":"767","DOI":"10.1109\/TRO.2005.844689","article-title":"A learning approach to robotic table tennis","volume":"21","author":"matsushima","year":"2005","journal-title":"IEEE Transactions on Robotics"},{"key":"ref5","doi-asserted-by":"crossref","DOI":"10.1609\/aaai.v25i1.8051","article-title":"Modeling opponent actions for table-tennis playing robot","author":"wang","year":"2011","journal-title":"AAAI"},{"key":"ref8","volume":"1","author":"sutton","year":"1998","journal-title":"Reinforcement Learning An Introduction"},{"key":"ref7","author":"lillicrap","year":"2015","journal-title":"Continuous control with deep reinforcement learning"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/nature16961"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1162\/089976600300015961"},{"key":"ref20","first-page":"1008","article-title":"Actor-critic algorithms","author":"konda","year":"2000","journal-title":"Advances in neural information processing systems"}],"event":{"name":"2018 IEEE International Conference on Real-time Computing and Robotics (RCAR)","location":"Kandima, Maldives","start":{"date-parts":[[2018,8,1]]},"end":{"date-parts":[[2018,8,5]]}},"container-title":["2018 IEEE International Conference on Real-time Computing and Robotics (RCAR)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8605389\/8621626\/08621776.pdf?arnumber=8621776","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,11]],"date-time":"2024-12-11T00:21:39Z","timestamp":1733876499000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8621776\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,8]]},"references-count":20,"URL":"https:\/\/doi.org\/10.1109\/rcar.2018.8621776","relation":{},"subject":[],"published":{"date-parts":[[2018,8]]}}}