{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T14:17:13Z","timestamp":1761401833180,"version":"3.37.3"},"reference-count":25,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"9","license":[{"start":{"date-parts":[[2018,9,1]],"date-time":"2018-09-01T00:00:00Z","timestamp":1535760000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Syst. Man Cybern, Syst."],"published-print":{"date-parts":[[2018,9]]},"DOI":"10.1109\/tsmc.2017.2671848","type":"journal-article","created":{"date-parts":[[2017,3,7]],"date-time":"2017-03-07T19:28:44Z","timestamp":1488914924000},"page":"1470-1481","source":"Crossref","is-referenced-by-count":8,"title":["Model Learning for Multistep Backward Prediction in Dyna-${Q}$  Learning"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9234-4836","authenticated-orcid":false,"given":"Kao-Shing","family":"Hwang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei-Cheng","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu-Jen","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Iris","family":"Hwang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCA.2012.2183349"},{"key":"ref11","first-page":"283","article-title":"Extended Dyna-Q algorithm for path planning of mobile robots","volume":"2","author":"viet","year":"2011","journal-title":"Meas Sci Instrum"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.2006.1642157"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2015.2418321"},{"key":"ref14","first-page":"18","article-title":"An empirical comparison of abstraction in models of Markov decision processes","author":"hester","year":"2009","journal-title":"Proceedings of the ICML\/UAI\/COLT Workshop on Abstraction in Reinforcement Learning"},{"key":"ref15","article-title":"Learning and using models","author":"hester","year":"2011","journal-title":"Reinforcement Learning State of the Art"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCA.2012.2227719"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2014.2358639"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1613\/jair.301"},{"article-title":"Learning from delayed rewards","year":"1989","author":"watkins","key":"ref19"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-141-3.50030-4"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2013.2294155"},{"key":"ref6","first-page":"717","article-title":"Generalized model learning for reinforcement learning in factored domains","author":"hester","year":"2009","journal-title":"Proc 8th Int Conf Auton Agents Multiagent Syst"},{"key":"ref5","first-page":"101","article-title":"Model-based reinforcement learning with an approximate, learned model","author":"kuvayev","year":"1996","journal-title":"Proc 9th Yale Workshop on Adaptive and Learning Syst"},{"key":"ref8","first-page":"1807","article-title":"A fast learning agent based on the Dyna architecture","volume":"30","author":"hsu","year":"2014","journal-title":"J Inf Sci Eng"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.2010.5509181"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2014.2373336"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCB.2008.921000"},{"journal-title":"Reinforcement Learning An Introduction","year":"1998","author":"sutton","key":"ref1"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2011.02.017"},{"key":"ref22","first-page":"70","article-title":"Decision tree function approximation in reinforcement learning","author":"pyeatt","year":"2001","journal-title":"Proc 3rd Int Symp Adapt Syst Evol Comput Probabilist Graph Models"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992698"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/BF00993104"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICNN.1993.298551"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/AERO.2005.1559688"}],"container-title":["IEEE Transactions on Systems, Man, and Cybernetics: Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6221021\/8438341\/07873361.pdf?arnumber=7873361","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,12]],"date-time":"2022-01-12T16:28:41Z","timestamp":1642004921000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/7873361\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,9]]},"references-count":25,"journal-issue":{"issue":"9"},"URL":"https:\/\/doi.org\/10.1109\/tsmc.2017.2671848","relation":{},"ISSN":["2168-2216","2168-2232"],"issn-type":[{"type":"print","value":"2168-2216"},{"type":"electronic","value":"2168-2232"}],"subject":[],"published":{"date-parts":[[2018,9]]}}}