{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,13]],"date-time":"2025-10-13T15:33:14Z","timestamp":1760369594694,"version":"3.28.0"},"reference-count":23,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1109\/cdc40024.2019.9029214","type":"proceedings-article","created":{"date-parts":[[2020,3,13]],"date-time":"2020-03-13T00:43:11Z","timestamp":1584060191000},"page":"6793-6798","source":"Crossref","is-referenced-by-count":2,"title":["Networked Control of Nonlinear Systems under Partial Observation Using Continuous Deep Q-Learning"],"prefix":"10.1109","author":[{"given":"Junya","family":"Ikemoto","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Toshimitsu","family":"Ushio","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","first-page":"387","article-title":"Deterministic Policy Gradient Algorithms","volume":"32","author":"silver","year":"2014","journal-title":"Proc 31st ICML"},{"key":"ref11","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-Level Control through Deep Reinforcement Learning","volume":"518","author":"mnih","year":"2015","journal-title":"Nature"},{"article-title":"Continuous Control with Deep Reinforcement Learning","year":"2016","author":"lillicrap","key":"ref12"},{"key":"ref13","first-page":"2829","article-title":"Continuous Deep QLearning with Model-based Acceleration","author":"gu","year":"2016","journal-title":"Proc 33rd ICML"},{"key":"ref14","first-page":"1928","article-title":"Asynchronous Methods for Deep Reinforcement Learning","author":"mnih","year":"2016","journal-title":"Proc 33rd ICML"},{"key":"ref15","first-page":"26","article-title":"Control of Nonholonomic Vehicle System Using Hierarchical Deep Reinforcement Learning","author":"masuda","year":"2017","journal-title":"Proc NOLTA2017"},{"key":"ref16","first-page":"5969","article-title":"Deep Reinforcement Learning Based Self-Conguring Integral Sliding Mode Control Scheme for Robot Manipulators","author":"sangiovanni","year":"2018","journal-title":"Proc of IEEE CDC 2018"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/LCSYS.2018.2847721"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2018.8619335"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2012.6425820"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCB.2010.2043839"},{"article-title":"Reinforcement Learning and Approximate Dynamic Programming for Feedback Control","year":"2013","author":"lewis","key":"ref3"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1587\/transfun.E99.A.454"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ECC.2015.7330910"},{"key":"ref8","first-page":"548","article-title":"Learning an Optimal Control Policy for a Markov Decision Process Under Linear Temporal Logic Specications","author":"hiromoto","year":"2015","journal-title":"Proc of IEEE Symposium Series on Computational Intelligence"},{"key":"ref7","first-page":"3372","article-title":"Robust Control of Uncertain Markov Decision Processes with Temporal Logic Specications","author":"wolff","year":"2012","journal-title":"Proc 2012 IEEE CDC"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/MCAS.2009.933854"},{"key":"ref1","article-title":"Reinforcement Learning: An Introduction","author":"sutton","year":"1999","journal-title":"A Bradford Book"},{"key":"ref9","first-page":"1057","article-title":"Policy Gradient Methods for Reinforcement Learning with Function Approximation","author":"sutton","year":"1999","journal-title":"Proc of the 12th NIPS"},{"key":"ref20","article-title":"Application of Continuous Deep Q-Learning to Networked State-Feedback Control of Nonlinear Systems with Uncertain Network Delays","author":"ikemoto","year":"2019","journal-title":"Proc NOLTA2019"},{"article-title":"ADAM: A Method for Stochastic Operation","year":"2014","author":"kingma","key":"ref22"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ACC.2005.1470171"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1103\/PhysRev.36.823"}],"event":{"name":"2019 IEEE 58th Conference on Decision and Control (CDC)","start":{"date-parts":[[2019,12,11]]},"location":"Nice, France","end":{"date-parts":[[2019,12,13]]}},"container-title":["2019 IEEE 58th Conference on Decision and Control (CDC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8977134\/9028853\/09029214.pdf?arnumber=9029214","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,19]],"date-time":"2022-07-19T16:17:47Z","timestamp":1658247467000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9029214\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,12]]},"references-count":23,"URL":"https:\/\/doi.org\/10.1109\/cdc40024.2019.9029214","relation":{},"subject":[],"published":{"date-parts":[[2019,12]]}}}