{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,30]],"date-time":"2026-01-30T09:49:52Z","timestamp":1769766592453,"version":"3.49.0"},"reference-count":42,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Fundamental Research Program 401 of Guangdong, China","award":["2020B1515310023"],"award-info":[{"award-number":["2020B1515310023"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2023]]},"DOI":"10.1109\/access.2023.3287098","type":"journal-article","created":{"date-parts":[[2023,6,16]],"date-time":"2023-06-16T17:30:39Z","timestamp":1686936639000},"page":"60292-60304","source":"Crossref","is-referenced-by-count":6,"title":["Neurodynamics Adaptive Reward and Action for Hand-to-Eye Calibration With Deep Reinforcement Learning"],"prefix":"10.1109","volume":"11","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5415-8338","authenticated-orcid":false,"given":"Zheng","family":"Zheng","sequence":"first","affiliation":[{"name":"School of Mathematics, South China University of Technology, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mengfei","family":"Yu","sequence":"additional","affiliation":[{"name":"School of Mathematics, South China University of Technology, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7662-9070","authenticated-orcid":false,"given":"Pengfei","family":"Guo","sequence":"additional","affiliation":[{"name":"School of Computational Science, Zhongkai University of Agriculture and Engineering, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7322-1873","authenticated-orcid":false,"given":"Delu","family":"Zeng","sequence":"additional","affiliation":[{"name":"School of Electronic and Information Engineering, South China University of Technology, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","article-title":"Playing Atari with deep reinforcement learning","author":"mnih","year":"2013","journal-title":"arXiv 1312 5602"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2017.2764529"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1512\/iumj.1957.6.56038"},{"key":"ref34","first-page":"278","article-title":"Policy invariance under reward transformations: Theory and application to reward shaping","author":"ng","year":"1999","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref15","first-page":"387","article-title":"Deterministic policy gradient algorithms","author":"silver","year":"2014","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2001.973374"},{"key":"ref14","first-page":"1057","article-title":"Policy gradient methods for reinforcement learning with function approximation","volume":"12","author":"sutton","year":"2000","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2018.2818747"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2018.2858744"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1177\/027836499501400301"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.3390\/math11071605"},{"key":"ref33","first-page":"1","article-title":"Actor-critic algorithms","author":"konda","year":"1999","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i6.20576"},{"key":"ref32","doi-asserted-by":"crossref","first-page":"354","DOI":"10.1038\/nature24270","article-title":"Mastering the game of go without human knowledge","volume":"550","author":"silver","year":"2017","journal-title":"Nature"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.promfg.2018.02.058"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2011.5980233"},{"key":"ref17","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","author":"fujimoto","year":"2018","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1002\/(SICI)1097-024X(199606)26:6<635::AID-SPE26>3.0.CO;2-P"},{"key":"ref16","article-title":"Continuous control with deep reinforcement learning","author":"lillicrap","year":"2015","journal-title":"arXiv 1509 02971"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2013.6696520"},{"key":"ref19","first-page":"1","article-title":"Hindsight experience replay","volume":"30","author":"andrychowicz","year":"2017","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref18","article-title":"Proximal policy optimization algorithms","author":"schulman","year":"2017","journal-title":"arXiv 1707 06347"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/70.326576"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/70.897783"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TIM.2020.2967958"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1177\/02783649922066213"},{"key":"ref20","first-page":"1889","article-title":"Trust region policy optimization","author":"schulman","year":"2015","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref42","article-title":"Soft actor-critic for discrete action settings","author":"christodoulou","year":"2019","journal-title":"arXiv 1910 07207"},{"key":"ref41","doi-asserted-by":"crossref","first-page":"3521","DOI":"10.1073\/pnas.1611835114","article-title":"Overcoming catastrophic forgetting in neural networks","volume":"114","author":"james","year":"2017","journal-title":"Proc Nat Acad Sci USA"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/s00138-017-0841-7"},{"key":"ref21","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","author":"haarnoja","year":"2018","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TIE.2020.2973893"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TIM.2019.2917235"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2011.5979569"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/70.88014"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.2010.5509415"},{"key":"ref9","first-page":"3","article-title":"Supervised machine learning: A review of classification techniques","volume":"160","author":"kotsiantis","year":"2007","journal-title":"Emerg Artif Intell Appl Comput Eng"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TIE.2013.2261034"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1117\/12.778135"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/j.procir.2014.06.043"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.aan5074"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/70.34770"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/10005208\/10154063.pdf?arnumber=10154063","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,10]],"date-time":"2023-07-10T19:33:39Z","timestamp":1689017619000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10154063\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"references-count":42,"URL":"https:\/\/doi.org\/10.1109\/access.2023.3287098","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023]]}}}