{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:23:43Z","timestamp":1740122623432,"version":"3.37.3"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2019,5,3]],"date-time":"2019-05-03T00:00:00Z","timestamp":1556841600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61671175"],"award-info":[{"award-number":["61671175"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61672190"],"award-info":[{"award-number":["61672190"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100011335","name":"Key Laboratory of Research on Chemistry and Physics of Optoelectronic Materials","doi-asserted-by":"crossref","award":["LabSOMP-2018-01"],"award-info":[{"award-number":["LabSOMP-2018-01"]}],"id":[{"id":"10.13039\/501100011335","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2019,10]]},"DOI":"10.1007\/s10489-019-01484-7","type":"journal-article","created":{"date-parts":[[2019,5,3]],"date-time":"2019-05-03T00:03:16Z","timestamp":1556841796000},"page":"3749-3764","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["An exploratory rollout policy for imagination-augmented agents"],"prefix":"10.1007","volume":"49","author":[{"given":"Peng","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yingnan","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xianglong","family":"Tang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zichan","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,5,3]]},"reference":[{"key":"1484_CR1","unstructured":"Andrychowicz M, Crow D, Ray A, Schneider J, Fong R, Welinder P, McGrew B, Tobin J, Abbeel P, Zaremba W (2017) Hindsight experience replay. In: Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, Long Beach, pp 5055\u20135065"},{"key":"1484_CR2","unstructured":"Bellemare MG, Dabney W, Munos R (2017) A distributional perspective on reinforcement learning. In: Proceedings of the 34th International Conference on Machine Learning, ICML 2017, Sydney, pp 449\u2013458"},{"issue":"1","key":"1484_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TCIAIG.2012.2186810","volume":"4","author":"C Browne","year":"2012","unstructured":"Browne C, Powley EJ, Whitehouse D, Lucas SM, Cowling PI, Rohlfshagen P, Tavener S, Liebana DP, Samothrakis S, Colton S (2012) A survey of monte carlo tree search methods. IEEE Trans Comput Intellig AI Games 4(1):1\u201343","journal-title":"IEEE Trans Comput Intellig AI Games"},{"key":"1484_CR4","unstructured":"Chiappa S, Racani\u0117re S, Wierstra D, Mohamed S (2017) Recurrent environment simulators"},{"issue":"6","key":"1484_CR5","doi-asserted-by":"publisher","first-page":"1347","DOI":"10.1162\/089976602753712972","volume":"14","author":"K Doya","year":"2002","unstructured":"Doya K, Samejima K, Katagiri K, Kawato M (2002) Multiple model-based reinforcement learning. Neural Comput 14(6):1347\u20131369","journal-title":"Neural Comput"},{"key":"1484_CR6","unstructured":"Feinberg V, Wan A, Stoica I, Jordan MI, Gonzalez JE, Levine S (2018) Model-based value estimation for efficient model-free reinforcement learning"},{"key":"1484_CR7","doi-asserted-by":"crossref","unstructured":"Finn C, Levine S (2017) Deep visual foresight for planning robot motion. In: 2017 IEEE International Conference on Robotics and Automation, ICRA 2017, Singapore, pp 2786\u20132793","DOI":"10.1109\/ICRA.2017.7989324"},{"key":"1484_CR8","unstructured":"Fortunato M, Azar MG, Piot B, Menick J, Osband I, Graves A, Mnih V, Munos R, Hassabis D, Pietquin O, Blundell C, Legg S (2017) Noisy networks for exploration. CoRR arXiv: 1706.10295"},{"key":"1484_CR9","unstructured":"Guu K, Pasupat P, Liu EZ, Liang P (2017) From language to programs: Bridging reinforcement learning and maximum marginal likelihood. In: Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics, ACL 2017. Long Papers, Vancouver, Vol 1, pp 1051\u20131062"},{"key":"1484_CR10","unstructured":"Ha D, Schmidhuber J (2018) World models"},{"key":"1484_CR11","doi-asserted-by":"crossref","unstructured":"van Hasselt H, Guez A, Silver D (2016) Deep reinforcement learning with double q-learning. In: Proceedings of the Thirtieth AAAI Conference on Artificial Intelligence, Phoenix, pp 2094\u20132100","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"1484_CR12","doi-asserted-by":"crossref","unstructured":"Hessel M, Modayil J, van Hasselt H, Schaul T, Ostrovski G, Dabney W, Horgan D, Piot B, Azar MG, Silver D (2018) Rainbow: Combining improvements in deep reinforcement learning. In: Proceedings of the Thirty-Second AAAI Conference on Artificial Intelligence, (AAAI-18), the 30th innovative Applications of Artificial Intelligence (IAAI-18), and the 8th AAAI Symposium on Educational Advances in Artificial Intelligence (EAAI-18), New Orleans, pp 3215\u20133222","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"1484_CR13","doi-asserted-by":"publisher","first-page":"1238","DOI":"10.1177\/0278364913495721","volume":"32","author":"J Kober","year":"2013","unstructured":"Kober J, Bagnell JA, Peters J (2013) Reinforcement learning in robotics: A survey. I J Robot Res 32:1238\u20131274","journal-title":"I J Robot Res"},{"key":"1484_CR14","unstructured":"Konda V (2002) Actor-critic algorithms. Ph.D. thesis, Massachusetts Institute of Technology, Cambridge"},{"key":"1484_CR15","doi-asserted-by":"crossref","unstructured":"Lample G, Chaplot DS (2017) Playing FPS games with deep reinforcement learning. In: Proceedings of the Thirty-First AAAI Conference on Artificial Intelligence, San Francisco, pp 2140\u20132146","DOI":"10.1609\/aaai.v31i1.10827"},{"key":"1484_CR16","doi-asserted-by":"crossref","unstructured":"Lenz I, Knepper RA, Saxena A (2015) Deepmpc: Learning deep latent features for model predictive control. In: Robotics: Science and Systems XI. Sapienza University of Rome, Rome","DOI":"10.15607\/RSS.2015.XI.012"},{"key":"1484_CR17","unstructured":"Levine S, Koltun V (2013) Guided policy search. In: Proceedings of the 30th International Conference on Machine Learning, ICML 2013, Atlanta, pp 1\u20139"},{"key":"1484_CR18","unstructured":"Li Y (2017) Deep reinforcement learning: An overview"},{"key":"1484_CR19","unstructured":"Lillicrap TP, Hunt JJ, Pritzel A, Heess N, Erez T, Tassa Y, Silver D, Wierstra D (2015) Continuous control with deep reinforcement learning"},{"key":"1484_CR20","unstructured":"Michalski V, Memisevic R, Konda KR (2014) Modeling deep temporal dependencies with recurrent \u201dgrammar cells\u201d. In: Advances in Neural Information Processing Systems 27: Annual Conference on Neural Information Processing Systems 2014, Montreal, pp 1925\u20131933"},{"key":"1484_CR21","unstructured":"Mittelman R, Kuipers B, Savarese S, Lee H (2014) Structured recurrent temporal restricted boltzmann machines. In: Proceedings of the 31th International Conference on Machine Learning, ICML 2014, Beijing, pp 1647\u20131655"},{"key":"1484_CR22","unstructured":"Mnih V, Badia AP, Mirza M, Graves A, Lillicrap TP, Harley T, Silver D, Kavukcuoglu K (2016) Asynchronous methods for deep reinforcement learning. In: Proceedings of the 33nd International Conference on Machine Learning, ICML 2016, New York City, pp 1928\u20131937"},{"key":"1484_CR23","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Graves A, Antonoglou I, Wierstra D, Riedmiller MA (2013) Playing atari with deep reinforcement learning"},{"issue":"7540","key":"1484_CR24","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller MA, Fidjeland A, Ostrovski G, Petersen S, Beattie C, Sadik A, Antonoglou I, King H, Kumaran D, Wierstra D, Legg S, Hassabis D (2015) Human-level control through deep reinforcement learning. Nature 518(7540):529\u2013533","journal-title":"Nature"},{"key":"1484_CR25","unstructured":"Oh J, Guo X, Lee H, Lewis RL, Singh S (2015) Action-conditional video prediction using deep networks in atari games. In: Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015, Montreal, pp 2863\u20132871"},{"key":"1484_CR26","unstructured":"Oh J, Guo Y, Singh S, Lee H (2018) Self-imitation learning. In: Proceedings of the 35th International Conference on Machine Learning, ICML 2018, Stockholmsma\u0307ssan, Stockholm, pp 3875\u20133884"},{"key":"1484_CR27","unstructured":"Oh J, Singh S, Lee H (2017) Value prediction network. In: Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, Long Beach, pp 6120\u20136130"},{"key":"1484_CR28","doi-asserted-by":"crossref","unstructured":"Pathak D, Agrawal P, Efros AA, Darrell T (2017) Curiosity-driven exploration by self-supervised prediction. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition Workshops, CVPR Workshops, Honolulu, pp 488\u2013489","DOI":"10.1109\/CVPRW.2017.70"},{"key":"1484_CR29","unstructured":"Racani\u0117re S, Weber T (2017) Imagination-augmented agents for deep reinforcement learning. In: Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, Long Beach, pp 5694\u20135705"},{"key":"1484_CR30","unstructured":"Schaul T, Quan J, Antonoglou I, Silver D (2015) Prioritized experience replay. CoRR 1511.05952"},{"key":"1484_CR31","unstructured":"Silver D, van Hasselt H, Hessel M, Schaul T, Guez A, Harley T, Dulac-Arnold G, Reichert DP, Rabinowitz NC, Barreto A, Degris T (2017) The predictron: End-to-end learning and planning. In: Proceedings of the 34th International Conference on Machine Learning, ICML, Sydney, NSW, pp 3191\u20133199"},{"issue":"7587","key":"1484_CR32","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2016","unstructured":"Silver D, Huang A, Maddison CJ, Guez A, Sifre L, van den Driessche G, Schrittwieser J, Antonoglou I, Panneershelvam V, Lanctot M, Dieleman S, Grewe D, Nham J, Kalchbrenner N, Sutskever I, Lillicrap TP, Leach M, Kavukcuoglu K, Graepel T, Hassabis D (2016) Mastering the game of go with deep neural networks and tree search. Nature 529(7587):484\u2013489","journal-title":"Nature"},{"key":"1484_CR33","unstructured":"Silver D, Hubert T, Schrittwieser J, Antonoglou I, Lai M, Guez A, Lanctot M, Sifre L, Kumaran D, Graepel T, Lillicrap TP, Simonyan K, Hassabis D (2017) Mastering chess and shogi by self-play with a general reinforcement learning algorithm. CoRR 1712.01815"},{"key":"1484_CR34","unstructured":"Silver D, Lever G, Heess N, Degris T, Wierstra D, Riedmiller MA (2014) Deterministic policy gradient algorithms. In: Proceedings of the 31th International Conference on Machine Learning, ICML 2014, Beijing, pp 387\u2013395"},{"key":"1484_CR35","unstructured":"Srivastava N, Mansimov E, Salakhutdinov R (2015) Unsupervised learning of video representations using lstms. In: Proceedings of the 32nd International Conference on Machine Learning, ICML 2015, Lille, pp 843\u2013852"},{"key":"1484_CR36","unstructured":"Stadie BC, Levine S, Abbeel P (2015) Incentivizing exploration in reinforcement learning with deep predictive models. In: NIPS Workshop"},{"key":"1484_CR37","first-page":"9","volume":"3","author":"RS Sutton","year":"1988","unstructured":"Sutton RS (1988) Learning to predict by the methods of temporal differences. Mach Learn 3:9\u201344","journal-title":"Mach Learn"},{"issue":"4","key":"1484_CR38","doi-asserted-by":"publisher","first-page":"160","DOI":"10.1145\/122344.122377","volume":"2","author":"RS Sutton","year":"1991","unstructured":"Sutton RS (1991) Dyna, an integrated architecture for learning, planning, and reacting. SIGART Bullet 2 (4):160\u2013163","journal-title":"SIGART Bullet"},{"key":"1484_CR39","doi-asserted-by":"crossref","unstructured":"Sutton RS, Barto AG, Bach F et al (1998) Reinforcement learning: An introduction. MIT press, Cambridge","DOI":"10.1109\/TNN.1998.712192"},{"key":"1484_CR40","unstructured":"Sutton RS, McAllester DA, Singh S, Mansour Y (1999) Policy gradient methods for reinforcement learning with function approximation. In: Advances in Neural Information Processing Systems 12, [NIPS Conference, Denver], pp 1057\u20131063"},{"key":"1484_CR41","unstructured":"Tamar A, Levine S, Abbeel P, Wu Y, Thomas G (2016) Value iteration networks. In: Advances in Neural Information Processing Systems 29: Annual Conference on Neural Information Processing Systems 2016, Barcelona, pp 2146\u20132154"},{"issue":"2","key":"1484_CR42","doi-asserted-by":"publisher","first-page":"88","DOI":"10.3233\/ICG-1995-18207","volume":"18","author":"G Tesauro","year":"1995","unstructured":"Tesauro G (1995) Temporal difference learning and td-gammon. ICGA J 18(2):88","journal-title":"ICGA J"},{"key":"1484_CR43","unstructured":"Wang Z, Schaul T, Hessel M, van Hasselt H, Lanctot M, de Freitas N (2016) Dueling network architectures for deep reinforcement learning. In: Proceedings of the 33nd International Conference on Machine Learning, ICML 2016, New York City, pp 1995\u20132003"},{"key":"1484_CR44","first-page":"279","volume":"8","author":"CJCH Watkins","year":"1992","unstructured":"Watkins CJCH, Dayan P (1992) Technical note q-learning. Mach Learn 8:279\u2013292","journal-title":"Mach Learn"},{"key":"1484_CR45","unstructured":"Watter M, Springenberg JT, Boedecker J, Riedmiller MA (2015) Embed to control: A locally linear latent dynamics model for control from raw images. In: Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015, Montreal, pp 2746\u20132754"},{"key":"1484_CR46","first-page":"229","volume":"8","author":"RJ Williams","year":"1992","unstructured":"Williams RJ (1992) Simple statistical gradient-following algorithms for connectionist reinforcement learning. Mach Learn 8:229\u2013256","journal-title":"Mach Learn"},{"key":"1484_CR47","unstructured":"Xia Y, He D, Qin T, Wang L, Yu N, Liu TY, Ma WY (2016) Dual learning for machine translation. In: NIPS"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-019-01484-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10489-019-01484-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-019-01484-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,17]],"date-time":"2022-09-17T09:12:14Z","timestamp":1663405934000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10489-019-01484-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,5,3]]},"references-count":47,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2019,10]]}},"alternative-id":["1484"],"URL":"https:\/\/doi.org\/10.1007\/s10489-019-01484-7","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"type":"print","value":"0924-669X"},{"type":"electronic","value":"1573-7497"}],"subject":[],"published":{"date-parts":[[2019,5,3]]},"assertion":[{"value":"3 May 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}