{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,27]],"date-time":"2026-01-27T21:55:34Z","timestamp":1769550934355,"version":"3.49.0"},"reference-count":67,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"2","license":[{"start":{"date-parts":[[2023,6,1]],"date-time":"2023-06-01T00:00:00Z","timestamp":1685577600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,6,1]],"date-time":"2023-06-01T00:00:00Z","timestamp":1685577600000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,6,1]],"date-time":"2023-06-01T00:00:00Z","timestamp":1685577600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,6,1]],"date-time":"2023-06-01T00:00:00Z","timestamp":1685577600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Cogn. Dev. Syst."],"published-print":{"date-parts":[[2023,6]]},"DOI":"10.1109\/tcds.2021.3061121","type":"journal-article","created":{"date-parts":[[2021,2,23]],"date-time":"2021-02-23T00:02:33Z","timestamp":1614038553000},"page":"395-408","source":"Crossref","is-referenced-by-count":2,"title":["Efficient Dialog Policy Learning With Hindsight, User Modeling, and Adaptation"],"prefix":"10.1109","volume":"15","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4844-5792","authenticated-orcid":false,"given":"Keting","family":"Lu","sequence":"first","affiliation":[{"name":"Commercialization Recommending Researching Department, Baidu Inc., Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9725-7747","authenticated-orcid":false,"given":"Yan","family":"Cao","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8992-9286","authenticated-orcid":false,"given":"Xiaoping","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4110-8213","authenticated-orcid":false,"given":"Shiqi","family":"Zhang","sequence":"additional","affiliation":[{"name":"Department of Computer Science, State University of New York at Binghamton, Binghamton, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1613\/jair.713"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v29i1.9385"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-1230"},{"key":"ref4","first-page":"733","article-title":"End-to-end task-completion neural dialogue systems","volume-title":"Proc. 8th Int. Joint Conf. Natural Lang. Process. Vol. 1 Long Papers","author":"Li"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683033"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33017289"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2012.2225812"},{"key":"ref8","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.1997.658989"},{"key":"ref10","volume-title":"Playing atari with deep reinforcement learning","author":"Mnih","year":"2013"},{"key":"ref11","volume-title":"End-to-end LSTM-based dialog control optimized with supervised and reinforcement learning","author":"Williams","year":"2016"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-10-2585-3_8"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00274"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33012596"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1203"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d18-1416"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.sigdial-1.40"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.3115\/1614108.1614146"},{"key":"ref19","volume-title":"A user simulator for task-completion dialogues","author":"Li","year":"2016"},{"key":"ref20","volume-title":"Efficient exploration for dialogue policy learning with bbq networks & replay buffer spiking","author":"Lipton","year":"2016"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W16-3601"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/p17-1062"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/p17-1045"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8268975"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-1237"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref27","volume-title":"Massively parallel methods for deep reinforcement learning","author":"Nair","year":"2015"},{"key":"ref28","volume-title":"Sample efficient actor\u2013critic with experience replay","author":"Wang","year":"2016"},{"key":"ref29","volume-title":"Prioritized experience replay","author":"Schaul","year":"2015"},{"key":"ref30","volume-title":"Multi-batch experience replay for fast convergence of continuous action control","author":"Han","year":"2017"},{"key":"ref31","first-page":"5048","article-title":"Hindsight experience replay","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Andrychowicz"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2015.03.007"},{"key":"ref33","first-page":"278","article-title":"Policy invariance under reward transformations: Theory and application to reward shaping","volume-title":"Proc. ICML","author":"Ng"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W15-4655"},{"key":"ref35","first-page":"1107","article-title":"Least-squares policy iteration","volume":"4","author":"Lagoudakis","year":"2003","journal-title":"J. Mach. Learn. Res."},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2010-40"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/1966407.1966412"},{"key":"ref38","first-page":"1878","article-title":"Sample efficient on-line learning of optimal dialogue policies with kalman temporal differences","volume-title":"Proc. Int. Joint Conf. Artif. Intell.","author":"Pietquin"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2013.2282190"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11946"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W17-5518"},{"key":"ref42","volume-title":"Continuously learning neural dialogue management","author":"Su","year":"2016"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.3389\/fnbot.2019.00115"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2006.890271"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.3389\/fnbot.2018.00030"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.3389\/fnbot.2013.00022"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.70"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/DEVLRN.2017.8329785"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/TCDS.2016.2538961"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.3389\/fpsyg.2013.00800"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.3389\/fnbot.2013.00020"},{"key":"ref52","volume-title":"Learning efficient representation for intrinsic motivation","author":"Zhao","year":"2019"},{"key":"ref53","first-page":"4403","article-title":"LIIR: Learning individual intrinsic reward in multi-agent reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Du"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1016\/j.swevo.2020.100715"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-29678-2_5019"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2008.2012071"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1253"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4471-4923-1"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/N15-1092"},{"key":"ref60","first-page":"1633","article-title":"Transfer learning for reinforcement learning domains: A survey","volume":"10","author":"Taylor","year":"2009","journal-title":"J. Mach. Learn. Res."},{"key":"ref61","volume-title":"Incentivizing exploration in reinforcement learning with deep predictive models","author":"Stadie","year":"2015"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-4013"},{"key":"ref63","volume-title":"Never give up: Learning directed exploration strategies","author":"Badia","year":"2020"},{"key":"ref64","volume-title":"AutoEG: Automated experience grafting for off-policy deep reinforcement learning","author":"Lu","year":"2020"},{"key":"ref65","first-page":"394","article-title":"Vision-and-dialog navigation","volume-title":"Proc. Conf. Robot. Learn.","author":"Thomason"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00387"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.3389\/fnbot.2019.00115"}],"container-title":["IEEE Transactions on Cognitive and Developmental Systems"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielam\/7274989\/10146524\/9360657-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7274989\/10146524\/09360657.pdf?arnumber=9360657","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,9]],"date-time":"2024-01-09T23:56:13Z","timestamp":1704844573000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9360657\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6]]},"references-count":67,"journal-issue":{"issue":"2"},"URL":"https:\/\/doi.org\/10.1109\/tcds.2021.3061121","relation":{},"ISSN":["2379-8920","2379-8939"],"issn-type":[{"value":"2379-8920","type":"print"},{"value":"2379-8939","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,6]]}}}