{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,16]],"date-time":"2025-09-16T17:20:08Z","timestamp":1758043208770,"version":"3.44.0"},"reference-count":22,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,8,25]],"date-time":"2025-08-25T00:00:00Z","timestamp":1756080000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,8,25]],"date-time":"2025-08-25T00:00:00Z","timestamp":1756080000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["CPS-1851588,CPS-2227185,S&AS-1849198,SATC-2231651"],"award-info":[{"award-number":["CPS-1851588,CPS-2227185,S&AS-1849198,SATC-2231651"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,8,25]]},"DOI":"10.1109\/ccta53793.2025.11151453","type":"proceedings-article","created":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T17:29:23Z","timestamp":1757611763000},"page":"15-20","source":"Crossref","is-referenced-by-count":0,"title":["Actively Learning Reinforcement Learning: A Stochastic Optimal Control Approach"],"prefix":"10.1109","author":[{"given":"Mohammad S.","family":"Ramadan","sequence":"first","affiliation":[{"name":"Argonne National Laboratory,Mathematics and Computer Science Division,Lemont,IL,USA,60439"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mahmoud A.","family":"Hayajnh","sequence":"additional","affiliation":[{"name":"The Daniel Guggenheim School of Aerospace Engineering, Georgia Institute of Technology,GA,USA,30332-0150"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Michael T.","family":"Tolley","sequence":"additional","affiliation":[{"name":"University of California,Department of Mechanical and Aerospace Engineering,San Diego,CA,USA,92161"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kyriakos G.","family":"Vamvoudakis","sequence":"additional","affiliation":[{"name":"The Daniel Guggenheim School of Aerospace Engineering, Georgia Institute of Technology,GA,USA,30332-0150"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/tsmc.1982.4308806"},{"volume-title":"Introduction to stochastic control theory","year":"2012","author":"\u00c5str\u00f6m","key":"ref2"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/0898-1221(86)90052-0"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/0022-247X(71)90161-2"},{"key":"ref5","article-title":"Dynamic programming and optimal control: Volume I","volume":"1","author":"Bertsekas","year":"2012","journal-title":"Athena scientific"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4757-3437-9"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1023\/A:1008935410038"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11694"},{"key":"ref9","article-title":"Reinforcement learning algorithm for partially observable markov decision problems","volume":"7","author":"Jaakkola","year":"1994","journal-title":"Advances in neural information processing systems"},{"volume-title":"Stochastic processes and filtering theory","year":"2007","author":"Jazwinski","key":"ref10"},{"key":"ref11","doi-asserted-by":"crossref","DOI":"10.1137\/1.9781611974263","volume-title":"Stochastic systems: Estimation, identification, and adaptive control","author":"Kumar","year":"2015"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.32657\/10356\/90191"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1002\/acs.2414"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-control-053018-023825"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/9.754809"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4612-1818-0"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2010.10.013"},{"key":"ref18","first-page":"387","article-title":"Deterministic policy gradient algorithms","volume-title":"International conference on machine learning","author":"Silver","year":"2014"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1016\/0022-247X(65)90027-2"},{"volume-title":"Reinforcement learning: An introduction","year":"2018","author":"Sutton","key":"ref20"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.1973.1100238"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2012.09.018"}],"event":{"name":"2025 IEEE Conference on Control Technology and Applications (CCTA)","start":{"date-parts":[[2025,8,25]]},"location":"San Diego, CA, USA","end":{"date-parts":[[2025,8,27]]}},"container-title":["2025 IEEE Conference on Control Technology and Applications (CCTA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11151269\/11151271\/11151453.pdf?arnumber=11151453","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,12]],"date-time":"2025-09-12T04:57:42Z","timestamp":1757653062000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11151453\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,25]]},"references-count":22,"URL":"https:\/\/doi.org\/10.1109\/ccta53793.2025.11151453","relation":{},"subject":[],"published":{"date-parts":[[2025,8,25]]}}}