{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T16:32:34Z","timestamp":1783701154046,"version":"3.55.0"},"reference-count":49,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"6","license":[{"start":{"date-parts":[[2021,6,1]],"date-time":"2021-06-01T00:00:00Z","timestamp":1622505600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,6,1]],"date-time":"2021-06-01T00:00:00Z","timestamp":1622505600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,6,1]],"date-time":"2021-06-01T00:00:00Z","timestamp":1622505600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001447","name":"Energy Program, Nation Research Foundation, Prime Minister\u2019s Office, Singapore, administrated by the Energy Market Authority of Singapore","doi-asserted-by":"publisher","award":["NRF2017EWT-EP003-023"],"award-info":[{"award-number":["NRF2017EWT-EP003-023"]}],"id":[{"id":"10.13039\/501100001447","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Green Data Centre Research administrated by the Info-communications Media Development Authority","award":["NRF2015ENC-GDCR01001-003"],"award-info":[{"award-number":["NRF2015ENC-GDCR01001-003"]}]},{"DOI":"10.13039\/501100001381","name":"Behavioral Studies in the Energy, Water, Waste and Transportation Sector","doi-asserted-by":"publisher","award":["BSEWWT2017_2_06"],"award-info":[{"award-number":["BSEWWT2017_2_06"]}],"id":[{"id":"10.13039\/501100001381","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2021,6]]},"DOI":"10.1109\/tnnls.2020.3008249","type":"journal-article","created":{"date-parts":[[2020,8,31]],"date-time":"2020-08-31T20:50:34Z","timestamp":1598907034000},"page":"2758-2771","source":"Crossref","is-referenced-by-count":12,"title":["Intelligent Trainer for Dyna-Style Model-Based Deep Reinforcement Learning"],"prefix":"10.1109","volume":"32","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0315-1125","authenticated-orcid":false,"given":"Linsen","family":"Dong","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2375-1806","authenticated-orcid":false,"given":"Yuanlong","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xin","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2751-5114","authenticated-orcid":false,"given":"Yonggang","family":"Wen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kyle","family":"Guan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","article-title":"Model-ensemble trust-region policy optimization","author":"kurutach","year":"2018","journal-title":"Proc 6th Int Conf Learn Represent ICLR"},{"key":"ref38","article-title":"Model-based reinforcement learning via meta-policy optimization","author":"clavera","year":"2018","journal-title":"arXiv 1809 05214"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/TSG.2018.2879572"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2018.2859801"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2015.2490698"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2013.2247627"},{"key":"ref37","first-page":"195","article-title":"Uncertainty-driven imagination for continuous deep reinforcement learning","author":"kalweit","year":"2017","journal-title":"Proc Conf Robot Learn"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2016.2527501"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2016.2522401"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2017.2712561"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2015.2417170"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2014.2319577"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/MCI.2018.2840727"},{"key":"ref2","article-title":"Playing atari with deep reinforcement learning","author":"mnih","year":"2013","journal-title":"arXiv 1312 5602"},{"key":"ref1","author":"sutton","year":"2018","journal-title":"Reinforcement Learning An Introduction"},{"key":"ref20","year":"2018","journal-title":"Codebase of Berkeley Deep RL Course"},{"key":"ref22","author":"dhariwal","year":"2017","journal-title":"OpenAI Baselines"},{"key":"ref21","year":"2018","journal-title":"Codebase of Trust Region Policy Optimization of Pat-Coady"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.1983.6313077"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2016.2582849"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2013.2281663"},{"key":"ref25","volume":"17","author":"lewis","year":"2013","journal-title":"Reinforcement Learning and Approximate Dynamic Programming for Feedback Control"},{"key":"ref10","first-page":"5690","article-title":"Imagination-augmented agents for deep reinforcement learning","author":"racani\u00e8re","year":"2017","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref11","article-title":"Learning model-based planning from scratch","author":"pascanu","year":"2017","journal-title":"arXiv 1707 06170"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1016\/j.apenergy.2015.05.081"},{"key":"ref12","article-title":"Openai gym","author":"brockman","year":"2016"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2019.2927410"},{"key":"ref14","first-page":"465","article-title":"PILCO: A model-based and data-efficient approach to policy search","author":"deisenroth","year":"2011","journal-title":"Proc 28th Int Conf Mach Learn ICML"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2015.7138994"},{"key":"ref16","year":"2018","journal-title":"Codebase of Intelligent Trainer for Dyna-style Model-Based Reinforcement Learning"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"ref18","volume":"1","author":"sutton","year":"1998","journal-title":"Reinforcement Learning An Introduction"},{"key":"ref19","first-page":"4026","article-title":"Deep exploration via bootstrapped DQN","author":"osband","year":"2016","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref4","first-page":"1889","article-title":"Continuous control with deep reinforcement learning","author":"lillicrap","year":"2016","journal-title":"Proc 4th Int Conf Learn Represent ICLR"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1038\/nature16961"},{"key":"ref6","first-page":"1","article-title":"Deep reinforcement learning in parameterized action space","author":"hausknecht","year":"2016","journal-title":"Proc Int Conf Learn Represent (ICLR)"},{"key":"ref5","first-page":"1889","article-title":"Trust region policy optimization","author":"schulman","year":"2015","journal-title":"Int Conf Mach Learn"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2011.VII.008"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/122344.122377"},{"key":"ref49","first-page":"8356","article-title":"Nemo: Neuro-evolution with multiobjective optimization of deep neural network for speed and accuracy","author":"kim","year":"2017","journal-title":"Proc ICML AutoML Workshop"},{"key":"ref9","first-page":"3338","article-title":"Deep learning for real-time game play using offline monte-carlo tree search planning","author":"guo","year":"2014","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref46","first-page":"21","article-title":"A brief review of the chalearn automl challenge: Any-time any-dataset learning without human intervention","author":"guyon","year":"2016","journal-title":"Proc Workshop Autom Mach Learn"},{"key":"ref45","first-page":"703","article-title":"Combining model-based and model-free updates for trajectory-centric reinforcement learning","author":"chebotar","year":"2017","journal-title":"Proc 34th Int Conf Mach Learn"},{"key":"ref48","first-page":"8356","article-title":"Transfer learning with neural automl","author":"wong","year":"2018","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01234-2_48"},{"key":"ref42","first-page":"264","article-title":"Lipschitz continuity in model-based reinforcement learning","author":"asadi","year":"2018","journal-title":"Proc 35th Int Conf Mach Learn ICML"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8463189"},{"key":"ref44","first-page":"1","article-title":"Integrating sample-based planning and model-based reinforcement learning","author":"walsh","year":"2010","journal-title":"Proc 24th AAAI Conf Artif Intell AAAI"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1162\/089976602753712972"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5962385\/9445711\/09181494.pdf?arnumber=9181494","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T14:53:15Z","timestamp":1652194395000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9181494\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,6]]},"references-count":49,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2020.3008249","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,6]]}}}