{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T08:16:56Z","timestamp":1783153016187,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":40,"publisher":"ACM","funder":[{"name":"&ldquo;Pioneer&rdquo; and &ldquo;Leading Goose&rdquo; R&D Program of Zhejiang","award":["2024C01212"],"award-info":[{"award-number":["2024C01212"]}]},{"name":"Hangzhou Key R&D Program","award":["2024SZD1A33"],"award-info":[{"award-number":["2024SZD1A33"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3774904.3792155","type":"proceedings-article","created":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T13:28:36Z","timestamp":1777296516000},"page":"650-661","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["FeedGuard: Online Critic-Guided Reinforcement Learning with Privacy-Preserving Feedback for Recommendation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9343-5729","authenticated-orcid":false,"given":"Mengying","family":"Zhu","sequence":"first","affiliation":[{"name":"School of Software Technology, Zhejiang University, Ningbo, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2290-1247","authenticated-orcid":false,"given":"Feiyue","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Software Technology, Zhejiang University, Ningbo, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1197-195X","authenticated-orcid":false,"given":"Lifan","family":"Jiang","sequence":"additional","affiliation":[{"name":"School of Software Technology, Zhejiang University, Ningbo, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3418-711X","authenticated-orcid":false,"given":"Mengyuan","family":"Yang","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9531-0906","authenticated-orcid":false,"given":"Yangyang","family":"Wu","sequence":"additional","affiliation":[{"name":"School of Software Technology, Zhejiang University, Ningbo, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2080-3903","authenticated-orcid":false,"given":"Guanjie","family":"Cheng","sequence":"additional","affiliation":[{"name":"School of Software Technology, Zhejiang University, Ningbo, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5483-0366","authenticated-orcid":false,"given":"Xiaolin","family":"Zheng","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,12]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"308","article-title":"Deep learning with differential privacy","author":"Abadi Martin","year":"2016","unstructured":"Martin Abadi, Andy Chu, Ian Goodfellow, H Brendan McMahan, Ilya Mironov, Kunal Talwar, and Li Zhang. 2016. Deep learning with differential privacy. In Proc. of the ACM SIGSAC. 308-318.","journal-title":"Proc. of the ACM SIGSAC."},{"key":"e_1_3_2_1_2_1","volume-title":"Proc. of NeurIPS.","author":"Brandfonbrener David","year":"2022","unstructured":"David Brandfonbrener, Alberto Bietti, Jacob Buckman, Romain Laroche, and Joan Bruna. 2022. When does return-conditioned supervised learning work for offline reinforcement learning?. In Proc. of NeurIPS."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3018661.3018702"},{"key":"e_1_3_2_1_4_1","first-page":"1339","article-title":"Reinforcement Mechanism Design for e-commerce","author":"Cai Qingpeng","year":"2018","unstructured":"Qingpeng Cai, Aris Filos-Ratsikas, Pingzhong Tang, and Yiwei Zhang. 2018. Reinforcement Mechanism Design for e-commerce. In Proc. of WWW. 1339-1348.","journal-title":"Proc. of WWW."},{"key":"e_1_3_2_1_5_1","volume-title":"Decision transformer: Reinforcement learning via sequence modeling. Advances in neural information processing systems","author":"Chen Lili","year":"2021","unstructured":"Lili Chen, Kevin Lu, Aravind Rajeswaran, Kimin Lee, Aditya Grover, Misha Laskin, Pieter Abbeel, Aravind Srinivas, and Igor Mordatch. 2021. Decision transformer: Reinforcement learning via sequence modeling. Advances in neural information processing systems, Vol. 34 (2021), 15084-15097."},{"key":"e_1_3_2_1_6_1","first-page":"376","article-title":"Maximum-Entropy Regularized Decision Transformer with Reward Relabelling for Dynamic Recommendation","author":"Chen Xiaocong","year":"2024","unstructured":"Xiaocong Chen, Siyu Wang, and Lina Yao. 2024. Maximum-Entropy Regularized Decision Transformer with Reward Relabelling for Dynamic Recommendation. In Proc. of KDD. 376-384.","journal-title":"Proc. of KDD."},{"key":"e_1_3_2_1_7_1","first-page":"1","volume-title":"Proc. of SIGIR","volume":"56","author":"Deffayet Romain","year":"2023","unstructured":"Romain Deffayet, Thibaut Thonet, Jean-Michel Renders, and Maarten De Rijke. 2023. Offline evaluation for reinforcement learning-based recommendation: a critical issue and some alternatives. In Proc. of SIGIR, Vol. 56. ACM New York, NY, USA, 1-14."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/11761679_29"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1561\/0400000042"},{"key":"e_1_3_2_1_10_1","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","author":"Fujimoto Scott","year":"2018","unstructured":"Scott Fujimoto, Herke Hoof, and David Meger. 2018. Addressing function approximation error in actor-critic methods. In Proc. of ICML. PMLR, 1587-1596.","journal-title":"Proc. of ICML. PMLR"},{"key":"e_1_3_2_1_11_1","first-page":"238","article-title":"Alleviating matthew effect of offline reinforcement learning in interactive recommendation","author":"Gao Chongming","year":"2023","unstructured":"Chongming Gao, Kexin Huang, Jiawei Chen, Yuan Zhang, Biao Li, Peng Jiang, Shiqi Wang, Zhong Zhang, and Xiangnan He. 2023. Alleviating matthew effect of offline reinforcement learning in interactive recommendation. In Proc. of SIGIR. 238-248.","journal-title":"Proc. of SIGIR."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.69554\/TCFN5165"},{"key":"e_1_3_2_1_13_1","volume-title":"Fedformer: Contextual federation with attention in reinforcement learning. arXiv preprint arXiv:2205.13697","author":"Hebert Liam","year":"2022","unstructured":"Liam Hebert, Lukasz Golab, Pascal Poupart, and Robin Cohen. 2022. Fedformer: Contextual federation with attention in reinforcement learning. arXiv preprint arXiv:2205.13697 (2022)."},{"key":"e_1_3_2_1_14_1","volume-title":"Session-based Recommendations with Recurrent Neural Networks. arXiv preprint arXiv:1511.06939","author":"Hidasi B","year":"2015","unstructured":"B Hidasi. 2015. Session-based Recommendations with Recurrent Neural Networks. arXiv preprint arXiv:1511.06939 (2015)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3074221"},{"key":"e_1_3_2_1_16_1","first-page":"197","article-title":"Self-attentive sequential recommendation","author":"Kang Wang-Cheng","year":"2018","unstructured":"Wang-Cheng Kang and Julian McAuley. 2018. Self-attentive sequential recommendation. In Proc. of ICDM. IEEE, 197-206.","journal-title":"Proc. of ICDM. IEEE"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.14778\/3503585.3503598"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i4.25566"},{"key":"e_1_3_2_1_19_1","volume-title":"Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971","author":"Lillicrap Timothy P","year":"2015","unstructured":"Timothy P Lillicrap, Jonathan J Hunt, Alexander Pritzel, Nicolas Heess, Tom Erez, Yuval Tassa, David Silver, and Daan Wierstra. 2015. Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971 (2015)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3548456"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657829"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/2365952.2365971"},{"key":"e_1_3_2_1_23_1","volume-title":"Federated reinforcement learning: Techniques, applications, and open challenges. arXiv preprint arXiv:2108.11887","author":"Qi Jiaju","year":"2021","unstructured":"Jiaju Qi, Qihao Zhou, Lei Lei, and Kan Zheng. 2021. Federated reinforcement learning: Techniques, applications, and open challenges. arXiv preprint arXiv:2108.11887 (2021)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","first-page":"6154","DOI":"10.52202\/079017-0199","article-title":"Federated ensemble-directed offline reinforcement learning","volume":"37","author":"Rengarajan Desik","year":"2024","unstructured":"Desik Rengarajan, Nitin Ragothaman, Dileep Kalathil, and Srinivas Shakkottai. 2024. Federated ensemble-directed offline reinforcement learning. Advances in Neural Information Processing Systems, Vol. 37 (2024), 6154-6179.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_25_1","volume-title":"Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347","author":"Schulman John","year":"2017","unstructured":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov. 2017. Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33014902"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3357384.3357895"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i8.20825"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-57959-7"},{"key":"e_1_3_2_1_30_1","first-page":"1599","article-title":"Causal decision transformer for recommender systems via offline reinforcement learning","author":"Wang Siyu","year":"2023","unstructured":"Siyu Wang, Xiaocong Chen, Dietmar Jannach, and Lina Yao. 2023a. Causal decision transformer for recommender systems via offline reinforcement learning. In Proc. of SIGIR. 1599-1608.","journal-title":"Proc. of SIGIR."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.119457"},{"key":"e_1_3_2_1_32_1","first-page":"4100","article-title":"Surrogate for long-term user experience in recommender systems","author":"Wang Yuyan","year":"2022","unstructured":"Yuyan Wang, Mohit Sharma, Can Xu, Sriraj Badam, Qian Sun, Lee Richardson, Lisa Chung, Ed H Chi, and Minmin Chen. 2022. Surrogate for long-term user experience in recommender systems. In Proc. of KDD. 4100-4109.","journal-title":"Proc. of KDD."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i14.29499"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2023.3242734"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2021.106946"},{"key":"e_1_3_2_1_36_1","volume-title":"Nguyen Quoc Viet Hung, and Hongzhi Yin","author":"Yuan Wei","year":"2024","unstructured":"Wei Yuan, Chaoqun Yang, Guanhua Ye, Tong Chen, Nguyen Quoc Viet Hung, and Hongzhi Yin. 2024. FELLAS: Enhancing Federated Sequential Recommendation with LLM as External Services. ACM Transactions on Information Systems (2024)."},{"key":"e_1_3_2_1_37_1","first-page":"1141","article-title":"User retention-oriented recommendation with decision transformer","author":"Zhao Kesen","year":"2023","unstructured":"Kesen Zhao, Lixin Zou, Xiangyu Zhao, Maolin Wang, and Dawei Yin. 2023. User retention-oriented recommendation with decision transformer. In Proc. of WWW. 1141-1149.","journal-title":"Proc. of WWW."},{"key":"e_1_3_2_1_38_1","first-page":"27042","article-title":"Online decision transformer","author":"Zheng Qinqing","year":"2022","unstructured":"Qinqing Zheng, Amy Zhang, and Aditya Grover. 2022b. Online decision transformer. In Proc. of ICML. PMLR, 27042-27059.","journal-title":"Proc. of ICML. PMLR"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSS.2022.3170691"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1080\/01621459.2024.2310287"}],"event":{"name":"WWW '26: The ACM Web Conference 2026","location":"Dubai United Arab Emirates","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM Web Conference 2026"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774904.3792155","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T07:31:39Z","timestamp":1783150299000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774904.3792155"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":40,"alternative-id":["10.1145\/3774904.3792155","10.1145\/3774904"],"URL":"https:\/\/doi.org\/10.1145\/3774904.3792155","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-04-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}