{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:47:29Z","timestamp":1784137649622,"version":"3.55.0"},"reference-count":34,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/OAPA.html"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61772532"],"award-info":[{"award-number":["61772532"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2019]]},"DOI":"10.1109\/access.2019.2922706","type":"journal-article","created":{"date-parts":[[2019,6,13]],"date-time":"2019-06-13T19:42:13Z","timestamp":1560454933000},"page":"79446-79454","source":"Crossref","is-referenced-by-count":29,"title":["Stochastic Double Deep Q-Network"],"prefix":"10.1109","volume":"7","author":[{"given":"Pingli","family":"Lv","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5327-1088","authenticated-orcid":false,"given":"Xuesong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2022-9999","authenticated-orcid":false,"given":"Yuhu","family":"Cheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ziming","family":"Duan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref33","first-page":"240","article-title":"Averaged-DQN: Variance reduction and stabilization for deep reinforcement learning","volume":"1","author":"anschel","year":"2017","journal-title":"Proc ICML"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TAAI.2017.19"},{"key":"ref31","first-page":"1032","article-title":"Estimating maximum expected value through Gaussian approximation","volume":"48","author":"d\u2019eramo","year":"2016","journal-title":"Proc ICML"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1023\/A:1013689704352"},{"key":"ref34","first-page":"2433","article-title":"Episodic memory deep Q-learning","author":"lin","year":"2018","journal-title":"Proc IJCAI"},{"key":"ref10","article-title":"Continuous control with deep reinforcement learning","author":"lillicrap","year":"0"},{"key":"ref11","article-title":"Prioritized experience replay","author":"schaul","year":"0"},{"key":"ref12","first-page":"2850","article-title":"Asynchronous methods for deep reinforcement learning","volume":"4","author":"mnih","year":"2016","journal-title":"Proc ICML"},{"key":"ref13","article-title":"Deep primal-dual reinforcement learning: Accelerating actor-critic using bellman duality","author":"cho","year":"2017","journal-title":"ariv 1712 02467"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1126\/science.aam6960"},{"key":"ref15","doi-asserted-by":"crossref","first-page":"354","DOI":"10.1038\/nature24270","article-title":"Mastering the game of Go without human knowledge","volume":"550","author":"silver","year":"2017","journal-title":"Nature"},{"key":"ref16","doi-asserted-by":"crossref","first-page":"484","DOI":"10.1038\/nature16961","article-title":"Mastering the game of Go with deep neural networks and tree search","volume":"529","author":"silver","year":"2016","journal-title":"Nature"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2017.2761841"},{"key":"ref18","article-title":"Towards vision-based deep reinforcement learning for robotic motion control","author":"zhang","year":"0"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.286"},{"key":"ref28","first-page":"2613","article-title":"Double Q-learning","author":"hasselt","year":"2010","journal-title":"Proc NIPS"},{"key":"ref4","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"mnih","year":"2015","journal-title":"Nature"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2018.2872751"},{"key":"ref3","article-title":"Playing atari with deep reinforcement learning","author":"mnih","year":"0"},{"key":"ref6","first-page":"2010","article-title":"Deep reinforcement learning with double Q-learning","author":"van hasselt","year":"2016","journal-title":"Proc AAAI"},{"key":"ref29","first-page":"282","article-title":"Bandit based Monte&#x2013;Carlo planning","author":"kocsis","year":"2006","journal-title":"Machine Learning ECML"},{"key":"ref5","first-page":"1","article-title":"Deep learning of representations: Looking forward","author":"bengio","year":"2013","journal-title":"Proc SLS"},{"key":"ref8","first-page":"29","article-title":"Deep recurrent q-learning for partially observable MDPs","author":"hausknecht","year":"2015","journal-title":"Proc AAAI-SDMIA"},{"key":"ref7","first-page":"2939","article-title":"Dueling network architectures for deep reinforcement learning","volume":"4","author":"wang","year":"2016","journal-title":"Proc ICML"},{"key":"ref2","article-title":"Learning from delayed rewards","author":"watkins","year":"1989"},{"key":"ref9","article-title":"Learning to communicate to solve riddles with deep distributed recurrent Q-networks","author":"foerster","year":"2016","journal-title":"arXiv 1602 02672"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2016.2618926"},{"key":"ref22","article-title":"Deep reinforcement learning for traffic light control in vehicular networks","author":"liang","year":"2018","journal-title":"arXiv 1803 11115"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2018.2821369"},{"key":"ref24","article-title":"Generating text with deep reinforcement learning","author":"guo","year":"0"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2018.2856520"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2015.2509646"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638947"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/8600701\/08736298.pdf?arnumber=8736298","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,27]],"date-time":"2022-01-27T10:03:35Z","timestamp":1643277815000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8736298\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019]]},"references-count":34,"URL":"https:\/\/doi.org\/10.1109\/access.2019.2922706","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019]]}}}