{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T06:49:55Z","timestamp":1778827795115,"version":"3.51.4"},"reference-count":46,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"1","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2025,1]]},"DOI":"10.1109\/tnnls.2023.3329808","type":"journal-article","created":{"date-parts":[[2023,11,16]],"date-time":"2023-11-16T19:08:38Z","timestamp":1700161718000},"page":"1044-1055","source":"Crossref","is-referenced-by-count":5,"title":["Plug-and-Play Model-Agnostic Counterfactual Policy Synthesis for Deep Reinforcement Learning-Based Recommendation"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-8726-5277","authenticated-orcid":false,"given":"Siyu","family":"Wang","sequence":"first","affiliation":[{"name":"School of Computer Science and Engineering, The University of New South Wales, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8849-4943","authenticated-orcid":false,"given":"Xiaocong","family":"Chen","sequence":"additional","affiliation":[{"name":"Data61, CSIRO, Eveleigh, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Julian","family":"McAuley","sequence":"additional","affiliation":[{"name":"Computer Science Department, University of California at San Diego (UCSD), La Jolla, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3207-172X","authenticated-orcid":false,"given":"Sally","family":"Cripps","sequence":"additional","affiliation":[{"name":"Human Technology Institute, University of Technology Sydney, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4149-839X","authenticated-orcid":false,"given":"Lina","family":"Yao","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, The University of New South Wales, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1007\/978-0-387-85820-3_3"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2017.08.008"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015394"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.26599\/BDMA.2018.9020008"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220122"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN48605.2020.9207010"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3336191.3371801"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3532015"},{"key":"ref9","article-title":"Deep reinforcement learning in large discrete action spaces","author":"Dulac-Arnold","year":"2015","journal-title":"arXiv:1512.07679"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3178876.3185994"},{"key":"ref11","article-title":"A survey of deep reinforcement learning in recommender systems: A systematic review and future directions","author":"Chen","year":"2021","journal-title":"arXiv:2109.03540"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3482244"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3482305"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462908"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462855"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/tnn.1998.712192"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1093\/bjps\/axi148"},{"key":"ref18","volume-title":"Elements of Causal Inference: Foundations and Learning Algorithms","author":"Peters","year":"2017"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511803161.009"},{"key":"ref20","article-title":"Policy distillation","author":"Rusu","year":"2015","journal-title":"arXiv:1511.06295"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33014902"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/2827872"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/415"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/1060745.1060754"},{"key":"ref25","article-title":"Continuous control with deep reinforcement learning","author":"Lillicrap","year":"2015","journal-title":"arXiv:1509.02971"},{"key":"ref26","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref27","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref28","article-title":"Sample-efficient reinforcement learning via counterfactual-based data augmentation","author":"Lu","year":"2020","journal-title":"arXiv:2012.09092"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013312"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1145\/3331184.3331203"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/1282100.1282114"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219886"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401225"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3482347"},{"key":"ref35","first-page":"1670","article-title":"Recommendations as treatments: Debiasing learning and evaluation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Schnabel"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/3240323.3240360"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401083"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462875"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2022.3156066"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462962"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3512251"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3482420"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i03.5631"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.590"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00471"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01081"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5962385\/10832116\/10319776.pdf?arnumber=10319776","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,8]],"date-time":"2025-01-08T20:21:52Z","timestamp":1736367712000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10319776\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1]]},"references-count":46,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2023.3329808","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1]]}}}