{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,7]],"date-time":"2025-05-07T05:04:33Z","timestamp":1746594273055,"version":"3.37.3"},"reference-count":74,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Knowl. Data Eng."],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1109\/tkde.2023.3344727","type":"journal-article","created":{"date-parts":[[2023,12,20]],"date-time":"2023-12-20T19:47:49Z","timestamp":1703101669000},"page":"7569-7584","source":"Crossref","is-referenced-by-count":3,"title":["Acquiring New Knowledge Without Losing Old Ones for Effective Continual Dialogue Policy Learning"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6147-8310","authenticated-orcid":false,"given":"Huimin","family":"Wang","sequence":"first","affiliation":[{"name":"Jarvis Research Center, Tencent YouTu Lab, Tencent, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yunyan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Jarvis Research Center, Tencent YouTu Lab, Tencent, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1891-2246","authenticated-orcid":false,"given":"Yifan","family":"Yang","sequence":"additional","affiliation":[{"name":"Tencent AI Lab, Tencent, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yefeng","family":"Zheng","sequence":"additional","affiliation":[{"name":"Jarvis Research Center, Tencent YouTu Lab, Tencent, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kam-Fai","family":"Wong","sequence":"additional","affiliation":[{"name":"Department of System Engineering and Engineering Management, The Chinese University of Hong Kong, Central Ave, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1007\/978-94-009-7758-7_1"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.566"},{"article-title":"End-to-end task-completion neural dialogue systems","year":"2017","author":"Li","key":"ref3"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-1237"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.21437\/Eurospeech.1997-380"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2000.859189"},{"article-title":"Continuously learning neural dialogue management","year":"2016","author":"Su","key":"ref7"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/1966407.1966412"},{"article-title":"Efficient exploration for dialog policy learning with deep BBQ networks & replay buffer spiking","year":"2016","author":"Lipton","key":"ref9"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1203"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.354"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.621"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1016\/s0079-7421(08)60536-8"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2019.01.012"},{"key":"ref15","first-page":"5966","article-title":"Memory replay GANs: Learning to generate new categories without forgetting","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Wu"},{"key":"ref16","first-page":"348","article-title":"Experience replay for continual learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Rolnick"},{"key":"ref17","first-page":"3849","article-title":"Interpolated policy gradient: Merging on-policy and off-policy gradient estimation for deep reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Gu"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/463"},{"key":"ref20","first-page":"2990","article-title":"Continual learning with deep generative replay","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Shin"},{"key":"ref21","first-page":"13122","article-title":"Episodic memory in lifelong language learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"de Masson dAutume"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-020-17866-2"},{"key":"ref23","first-page":"6467","article-title":"Gradient episodic memory for continual learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Lopez-Paz"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1611835114"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.310"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-short.66"},{"article-title":"Invariant risk minimization","year":"2019","author":"Arjovsky","key":"ref27"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.21437\/ICSLP.1998-92"},{"issue":"13","key":"ref29","first-page":"212","article-title":"Research and implementation of frame-based dialogue management model","volume":"31","author":"Yuan","year":"2005","journal-title":"Comput. Eng."},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2005.1566518"},{"key":"ref31","first-page":"140","article-title":"Example-based spoken dialogue system using WOZ system log","volume-title":"Proc. 4th SIGdial Workshop Discourse Dialogue","author":"Murao"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICSMC.2001.969811"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.01.008"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2012.2225812"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1062"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-1237"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461918"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11946"},{"article-title":"Results of the multi-domain task-completion dialog challenge","volume-title":"Proc. 34th AAAI Conf. Artif. Intell., 8th Dialog Syst. Technol. Challenge Workshop","author":"Li","key":"ref39"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-3011"},{"key":"ref41","first-page":"3352","article-title":"Reinforcement learning from demonstration through shaping","volume-title":"Proc. 24th Int. Joint Conf. Artif. Intell.","author":"Brys"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/E17-2032"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1010"},{"article-title":"Guided dialog policy learning without adversarial learning in the loop","year":"2020","author":"Li","key":"ref44"},{"key":"ref45","first-page":"3366","article-title":"Policy shaping with human teachers","volume-title":"Proc. 24th Int. Joint Conf. Artif. Intell.","author":"Cederborg"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11757"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.129"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21320"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21421"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11595"},{"article-title":"Progressive neural networks","year":"2016","author":"Rusu","key":"ref51"},{"article-title":"Lifelong learning with dynamically expandable networks","year":"2017","author":"Yoon","key":"ref52"},{"key":"ref53","first-page":"266","article-title":"Dynamic dialogue policy for continual reinforcement learning","volume-title":"Proc. 29th Int. Conf. Comput. Linguistics","author":"Geishauser"},{"article-title":"Toward continual learning for conversational agents","year":"2017","author":"Lee","key":"ref54"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.590"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1998.674402"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1613\/jair.301"},{"article-title":"Novel methods for efficient dialogue policy learning by improving agent-user interaction","year":"2019","author":"Peng","key":"ref58"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/e17-1042"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1061"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W17-5518"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1010"},{"article-title":"Continual reinforcement learning with multi-timescale replay","year":"2020","author":"Kaplanis","key":"ref63"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1037\/0278-7393.16.5.927"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1111\/j.1467-9280.1991.tb00175.x"},{"key":"ref66","article-title":"Note on the power law of forgetting","author":"Kahana","year":"2017","journal-title":"bioRxiv"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1547"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00314"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W19-5932"},{"key":"ref70","first-page":"705","article-title":"CORA: Benchmarks, baselines, and metrics as a platform for continual reinforcement learning agents","volume-title":"Proc. Conf. Lifelong Learn. Agents","author":"Powers"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01252-6_33"},{"key":"ref72","first-page":"1407","article-title":"IMPALA: Scalable distributed Deep-RL with importance weighted actor-learner architectures","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Espeholt"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-demos.19"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/589"}],"container-title":["IEEE Transactions on Knowledge and Data Engineering"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/69\/10750897\/10366832.pdf?arnumber=10366832","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,26]],"date-time":"2024-11-26T23:45:55Z","timestamp":1732664755000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10366832\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12]]},"references-count":74,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tkde.2023.3344727","relation":{},"ISSN":["1041-4347","1558-2191","2326-3865"],"issn-type":[{"type":"print","value":"1041-4347"},{"type":"electronic","value":"1558-2191"},{"type":"electronic","value":"2326-3865"}],"subject":[],"published":{"date-parts":[[2024,12]]}}}