{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T02:20:46Z","timestamp":1785205246515,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,8,14]],"date-time":"2022-08-14T00:00:00Z","timestamp":1660435200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,8,14]]},"DOI":"10.1145\/3534678.3539040","type":"proceedings-article","created":{"date-parts":[[2022,8,12]],"date-time":"2022-08-12T19:06:12Z","timestamp":1660331172000},"page":"4510-4520","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":45,"title":["Multi-Task Fusion via Reinforcement Learning for Long-Term User Satisfaction in Recommender Systems"],"prefix":"10.1145","author":[{"given":"Qihua","family":"Zhang","sequence":"first","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junning","family":"Liu","sequence":"additional","affiliation":[{"name":"Tencent, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuzhuo","family":"Dai","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yiyan","family":"Qi","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yifan","family":"Yuan","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kunlun","family":"Zheng","sequence":"additional","affiliation":[{"name":"Tencent, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fan","family":"Huang","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xianfeng","family":"Tan","sequence":"additional","affiliation":[{"name":"Tencent, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,8,14]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Reinforcement learning based recommender systems: A survey. arXiv preprint arXiv:2101.06286","author":"Afsar M Mehdi","year":"2021","unstructured":"M Mehdi Afsar, Trafford Crump, and Behrouz Far. 2021. Reinforcement learning based recommender systems: A survey. arXiv preprint arXiv:2101.06286 (2021)."},{"key":"e_1_3_2_2_2_1","volume-title":"Evolution strategies--a comprehensive introduction. Natural computing","author":"Beyer Hans-Georg","year":"2002","unstructured":"Hans-Georg Beyer and Hans-Paul Schwefel. 2002. Evolution strategies--a comprehensive introduction. Natural computing, Vol. 1, 1 (2002), 3--52."},{"key":"e_1_3_2_2_3_1","volume-title":"Multi-Objective Ranking Optimization for Product Search Using Stochastic Label Aggregation. In WWW","author":"Carmel David","year":"2020","unstructured":"David Carmel, Elad Haramaty, Arnon Lazerson, and Liane Lewin-Eytan. 2020. Multi-Objective Ranking Optimization for Product Search Using Stochastic Label Aggregation. In WWW 2020. 373--383."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"crossref","unstructured":"Paul Covington Jay Adams and Emre Sargin. 2016. Deep neural networks for youtube recommendations. In RecSys. 191--198.","DOI":"10.1145\/2959100.2959190"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"crossref","unstructured":"Stephen Dankwa and Wenfeng Zheng. 2019. Twin-delayed DDPG: A deep reinforcement learning technique to model a continuous movement of an intelligent robot agent. In ICVISP. 1--5.","DOI":"10.1145\/3387168.3387199"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"crossref","unstructured":"Anlei Dong Yi Chang Zhaohui Zheng Gilad Mishne Jing Bai Ruiqiang Zhang Karolina Buchner Ciya Liao and Fernando Diaz. 2010. Towards recency ranking in web search. In WSDM. 11--20.","DOI":"10.1145\/1718487.1718490"},{"key":"e_1_3_2_2_7_1","volume-title":"Deep reinforcement learning in large discrete action spaces. arXiv preprint arXiv:1512.07679","author":"Dulac-Arnold Gabriel","year":"2015","unstructured":"Gabriel Dulac-Arnold, Richard Evans, Hado van Hasselt, Peter Sunehag, Timothy Lillicrap, Jonathan Hunt, Timothy Mann, Theophane Weber, Thomas Degris, and Ben Coppin. 2015. Deep reinforcement learning in large discrete action spaces. arXiv preprint arXiv:1512.07679 (2015)."},{"key":"e_1_3_2_2_8_1","unstructured":"Scott Fujimoto Herke Hoof and David Meger. 2018. Addressing function approximation error in actor-critic methods. In ICML. PMLR 1587--1596."},{"key":"e_1_3_2_2_9_1","volume-title":"ICML. PMLR","author":"Fujimoto Scott","year":"2019","unstructured":"Scott Fujimoto, David Meger, and Doina Precup. 2019. Off-policy deep reinforcement learning without exploration. In ICML. PMLR, 2052--2062."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10287-020-00376-3"},{"key":"e_1_3_2_2_11_1","unstructured":"Yulong Gu Zhuoye Ding Shuaiqiang Wang and Dawei Yin. 2020 a. Hierarchical User Profiling for E-commerce Recommender Systems. In WSDM. 223--231."},{"key":"e_1_3_2_2_12_1","unstructured":"Yulong Gu Zhuoye Ding Shuaiqiang Wang Lixin Zou Yiding Liu and Dawei Yin. 2020 b. Deep Multifaceted Transformers for Multi-objective Ranking in Large-Scale E-commerce Recommender Systems. In CIKM. 2493--2500."},{"key":"e_1_3_2_2_13_1","volume-title":"HLGPS: a home location global positioning system in location-based social networks","author":"Gu Yulong","unstructured":"Yulong Gu, Jiaxing Song, Weidong Liu, and Lixin Zou. 2016. HLGPS: a home location global positioning system in location-based social networks. In ICDM. IEEE, 901--906."},{"key":"e_1_3_2_2_14_1","volume-title":"International conference on machine learning. PMLR","author":"Haarnoja Tuomas","year":"2018","unstructured":"Tuomas Haarnoja, Aurick Zhou, Pieter Abbeel, and Sergey Levine. 2018. Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. In International conference on machine learning. PMLR, 1861--1870."},{"key":"e_1_3_2_2_15_1","volume-title":"Optimizing Ranking Algorithm in Recommender System via Deep Reinforcement Learning","author":"Han Jianhua","unstructured":"Jianhua Han, Yong Yu, Feng Liu, Ruiming Tang, and Yuzhou Zhang. 2019. Optimizing Ranking Algorithm in Recommender System via Deep Reinforcement Learning. In AIAM. IEEE, 22--26."},{"key":"e_1_3_2_2_16_1","unstructured":"Xinran He Junfeng Pan Ou Jin Tianbing Xu Bo Liu Tao Xu Yanxin Shi Antoine Atallah Ralf Herbrich Stuart Bowers et al. 2014. Practical lessons from predicting clicks on ads at facebook. In ADKDD. 1--9."},{"key":"e_1_3_2_2_17_1","volume-title":"Conservative q-learning for offline reinforcement learning. arXiv preprint arXiv:2006.04779","author":"Kumar Aviral","year":"2020","unstructured":"Aviral Kumar, Aurick Zhou, George Tucker, and Sergey Levine. 2020. Conservative q-learning for offline reinforcement learning. arXiv preprint arXiv:2006.04779 (2020)."},{"key":"e_1_3_2_2_18_1","volume-title":"International Conference on Machine Learning. PMLR, 3703--3712","author":"Le Hoang","year":"2019","unstructured":"Hoang Le, Cameron Voloshin, and Yisong Yue. 2019. Batch policy learning under constraints. In International Conference on Machine Learning. PMLR, 3703--3712."},{"key":"e_1_3_2_2_19_1","volume-title":"com recommendations: Item-to-item collaborative filtering","author":"Linden Greg","year":"2003","unstructured":"Greg Linden, Brent Smith, and Jeremy York. 2003. Amazon. com recommendations: Item-to-item collaborative filtering. IEEE Internet computing, Vol. 7, 1 (2003), 76--80."},{"key":"e_1_3_2_2_20_1","volume-title":"H Chi","author":"Ma Jiaqi","year":"2018","unstructured":"Jiaqi Ma, Zhe Zhao, Xinyang Yi, Jilin Chen, Lichan Hong, and Ed H Chi. 2018. Modeling task relationships in multi-task learning with multi-gate mixture-of-experts. In SIGKDD. 1930--1939."},{"key":"e_1_3_2_2_21_1","volume-title":"An introduction to genetic algorithms","author":"Mitchell Melanie","unstructured":"Melanie Mitchell. 1998. An introduction to genetic algorithms .MIT press."},{"key":"e_1_3_2_2_22_1","volume-title":"Optimization techniques IFIP technical conference","author":"Jonas Movc","unstructured":"Jonas Movc kus. 1975. On Bayesian methods for seeking the extremum. In Optimization techniques IFIP technical conference. Springer, 400--404."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"crossref","unstructured":"Wentao Ouyang Xiuwu Zhang Li Li Heng Zou Xin Xing Zhaojie Liu and Yanlong Du. 2019. Deep spatio-temporal neural networks for click-through rate prediction. In SIGKDD. 2078--2086.","DOI":"10.1145\/3292500.3330655"},{"key":"e_1_3_2_2_24_1","volume-title":"Value-aware recommendation based on reinforced profit maximization in e-commerce systems. arXiv preprint arXiv:1902.00851","author":"Pei Changhua","year":"2019","unstructured":"Changhua Pei, Xinru Yang, Qing Cui, Xiao Lin, Fei Sun, Peng Jiang, Wenwu Ou, and Yongfeng Zhang. 2019. Value-aware recommendation based on reinforced profit maximization in e-commerce systems. arXiv preprint arXiv:1902.00851 (2019)."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"crossref","unstructured":"Qi Pi Weijie Bian Guorui Zhou Xiaoqiang Zhu and Kun Gai. 2019. Practice on long sequential user behavior modeling for click-through rate prediction. In SIGKDD. 2671--2679.","DOI":"10.1145\/3292500.3330666"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/2629350"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"crossref","unstructured":"Mario Rodriguez Christian Posse and Ethan Zhang. 2012. Multiple objective optimization in recommender systems. In RecSys. 11--18.","DOI":"10.1145\/2365952.2365961"},{"key":"e_1_3_2_2_28_1","unstructured":"David Silver Guy Lever Nicolas Heess Thomas Degris Daan Wierstra and Martin Riedmiller. 2014. Deterministic policy gradient algorithms. In ICML. PMLR 387--395."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"crossref","unstructured":"Hongyan Tang Junning Liu Ming Zhao and Xudong Gong. 2020. Progressive layered extraction (ple): A novel multi-task learning (mtl) model for personalized recommendations. In RecSys. 269--278.","DOI":"10.1145\/3383313.3412236"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1609\/aimag.v21i1.1501"},{"key":"e_1_3_2_2_31_1","volume-title":"Empirical study of off-policy policy evaluation for reinforcement learning. arXiv preprint arXiv:1911.06854","author":"Voloshin Cameron","year":"2019","unstructured":"Cameron Voloshin, Hoang M Le, Nan Jiang, and Yisong Yue. 2019. Empirical study of off-policy policy evaluation for reinforcement learning. arXiv preprint arXiv:1911.06854 (2019)."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"crossref","unstructured":"Lu Wang Wei Zhang Xiaofeng He and Hongyuan Zha. 2018. Supervised reinforcement learning with recurrent neural network for dynamic treatment recommendation. In SIGKDD. 2447--2456.","DOI":"10.1145\/3219819.3219961"},{"key":"e_1_3_2_2_33_1","volume-title":"Uncertainty Weighted Actor-Critic for Offline Reinforcement Learning. arXiv preprint arXiv:2105.08140","author":"Wu Yue","year":"2021","unstructured":"Yue Wu, Shuangfei Zhai, Nitish Srivastava, Joshua Susskind, Jian Zhang, Ruslan Salakhutdinov, and Hanlin Goh. 2021. Uncertainty Weighted Actor-Critic for Offline Reinforcement Learning. arXiv preprint arXiv:2105.08140 (2021)."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"crossref","unstructured":"Xiangyu Zhao Long Xia Liang Zhang Zhuoye Ding Dawei Yin and Jiliang Tang. 2018. Deep reinforcement learning for page-wise recommendations. In RecSys. 95--103.","DOI":"10.1145\/3240323.3240374"},{"key":"e_1_3_2_2_35_1","volume-title":"Chi","author":"Zhao Zhe","year":"2019","unstructured":"Zhe Zhao, Lichan Hong, Li Wei, Jilin Chen, Aniruddh Nath, Shawn Andrews, Aditee Kumthekar, Maheswaran Sathiamoorthy, Xinyang Yi, and Ed Chi. 2019. Recommending what video to watch next: a multitask ranking system. In RecSys. 43--51."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33015941"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"crossref","unstructured":"Guorui Zhou Xiaoqiang Zhu Chenru Song Ying Fan Han Zhu Xiao Ma Yanghui Yan Junqi Jin Han Li and Kun Gai. 2018. Deep interest network for click-through rate prediction. In SIGKDD. 1059--1068.","DOI":"10.1145\/3219819.3219823"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"crossref","unstructured":"Lixin Zou Long Xia Zhuoye Ding Jiaxing Song Weidong Liu and Dawei Yin. 2019. Reinforcement learning to optimize long-term user engagement in recommender systems. In SIGKDD. 2810--2818.","DOI":"10.1145\/3292500.3330668"}],"event":{"name":"KDD '22: The 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Washington DC USA","acronym":"KDD '22","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3534678.3539040","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3534678.3539040","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T18:09:50Z","timestamp":1750183790000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3534678.3539040"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,14]]},"references-count":38,"alternative-id":["10.1145\/3534678.3539040","10.1145\/3534678"],"URL":"https:\/\/doi.org\/10.1145\/3534678.3539040","relation":{},"subject":[],"published":{"date-parts":[[2022,8,14]]},"assertion":[{"value":"2022-08-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}