{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,20]],"date-time":"2025-10-20T18:40:52Z","timestamp":1760985652909,"version":"3.37.3"},"reference-count":43,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"6","license":[{"start":{"date-parts":[[2018,6,1]],"date-time":"2018-06-01T00:00:00Z","timestamp":1527811200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"}],"funder":[{"DOI":"10.13039\/100000006","name":"U.S. Office of Naval Research","doi-asserted-by":"publisher","award":["N00014-15-1-2103","N00014-14-1-0542"],"award-info":[{"award-number":["N00014-15-1-2103","N00014-14-1-0542"]}],"id":[{"id":"10.13039\/100000006","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100007698","name":"University of Florida Research Fellowship","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100007698","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Robert C. Pittman Research Fellowship"},{"name":"ASEE Naval Research Enterprise Fellowship"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2018,6]]},"DOI":"10.1109\/tnnls.2018.2812709","type":"journal-article","created":{"date-parts":[[2018,3,27]],"date-time":"2018-03-27T19:02:50Z","timestamp":1522177370000},"page":"2080-2098","source":"Crossref","is-referenced-by-count":21,"title":["Guided Policy Exploration for Markov Decision Processes Using an Uncertainty-Based Value-of-Information Criterion"],"prefix":"10.1109","volume":"29","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7755-5886","authenticated-orcid":false,"given":"Isaac J.","family":"Sledge","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Matthew S.","family":"Emigh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jose C.","family":"Principe","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1016\/0004-3702(94)00011-O"},{"key":"ref38","first-page":"1729","article-title":"A Bayesian approach for learning and planning in partially observable Markov decision processes","volume":"12","author":"ross","year":"2011","journal-title":"J Mach Learn Res"},{"key":"ref33","first-page":"751","article-title":"Gaussian processes in reinforcement learning","author":"kuss","year":"2003","journal-title":"Advances in neural information processing systems"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143932"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553441"},{"key":"ref30","first-page":"150","article-title":"Model based Bayesian exploration","author":"dearden","year":"1999","journal-title":"Proc Conf Uncertainty Artif Intell (UAI)"},{"key":"ref37","first-page":"476","article-title":"Model-based Bayesian reinforcement learning in large structured domains","author":"ross","year":"2008","journal-title":"Proc Conf Uncertainty Artif Intell (UAI)"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/1273496.1273534"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/1102351.1102377"},{"key":"ref34","first-page":"154","article-title":"Bayes meets Bellman: The Gaussian process approach to temporal difference learning","author":"engel","year":"2003","journal-title":"Proc Int Conf Mach Learn (ICML)"},{"key":"ref10","first-page":"3","article-title":"Value of information when an estimated random variable is hidden","volume":"3","author":"stratonovich","year":"1966","journal-title":"Izvestiya USSR Acad Sci Tech"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952670"},{"key":"ref11","first-page":"265","article-title":"Information theoretic learning","author":"principe","year":"2000","journal-title":"Unsupervised Adaptive Filtering"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TG.2018.2808201"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.3390\/e20030155"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/MLSP.2015.7324371"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-200-7.50075-1"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-141-3.50030-4"},{"key":"ref17","first-page":"740","article-title":"Efficient reinforcement learning in factored MDPs","author":"kearns","year":"1999","journal-title":"Proc Int Joint Conf Artif Intell (IJCAI)"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1023\/A:1017984413808"},{"key":"ref19","first-page":"695","article-title":"Convergence of indirect adaptive asynchronous value iteration algorithms","author":"gullapalli","year":"1993","journal-title":"Advances in neural information processing systems"},{"key":"ref28","first-page":"116","article-title":"Expected mistake bound model for on-line reinforcement learning","author":"fiechter","year":"1997","journal-title":"Proc Int Conf Mach Learn (ICML)"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2016.2585520"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1145\/180139.181019"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2016.2561300"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2017.2660070"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-335-6.50040-4"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2016.2609500"},{"key":"ref8","doi-asserted-by":"crossref","first-page":"237","DOI":"10.1613\/jair.301","article-title":"Reinforcement learning: A survey","volume":"4","author":"kaelbling","year":"1996","journal-title":"J Artif Intell Res"},{"key":"ref7","first-page":"531","article-title":"Active exploration in dynamic environments","author":"thrun","year":"1992","journal-title":"Advances in neural information processing systems"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TASE.2016.2517155"},{"key":"ref9","first-page":"3","article-title":"On value of information","volume":"5","author":"stratonovich","year":"1965","journal-title":"Izvestiya USSR Acad Sci Tech"},{"journal-title":"Reinforcement Learning An Introduction","year":"1998","author":"sutton","key":"ref1"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21236\/ADA276517"},{"key":"ref22","first-page":"213","article-title":"A general polynomial time algorithm for near-optimal reinforcement learning","volume":"3","author":"brafman","year":"2002","journal-title":"J Mach Learn Res"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1016\/S0004-3702(00)00039-4"},{"key":"ref42","first-page":"1","article-title":"Partitioning relational matrices of similarities or dissimilarities using the value of information","author":"sledge","year":"2018","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process (ICASSP)"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143955"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/18.144705"},{"key":"ref23","first-page":"485","article-title":"Incremental model-based learners with formal learning-time guarantees","author":"strehl","year":"2006","journal-title":"Proc Conf Uncertainty Artif Intell (UAI)"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/238061.238084"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2014.2369345"},{"key":"ref25","first-page":"2413","article-title":"Reinforcement learning in finite MDPs: PAC analysis","volume":"10","author":"strehl","year":"2009","journal-title":"J Mach Learn Res"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5962385\/8360119\/08326734.pdf?arnumber=8326734","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,12]],"date-time":"2022-01-12T16:12:34Z","timestamp":1642003954000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8326734\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,6]]},"references-count":43,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2018.2812709","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"type":"print","value":"2162-237X"},{"type":"electronic","value":"2162-2388"}],"subject":[],"published":{"date-parts":[[2018,6]]}}}