{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T05:44:55Z","timestamp":1777873495876,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,8,3]]},"DOI":"10.1145\/3711896.3737264","type":"proceedings-article","created":{"date-parts":[[2025,8,3]],"date-time":"2025-08-03T21:03:27Z","timestamp":1754255007000},"page":"4611-4622","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Supervised Learning-enhanced Multi-Group Actor Critic for Live Stream Allocation in Feed"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2526-8196","authenticated-orcid":false,"given":"Jingxin","family":"Liu","sequence":"first","affiliation":[{"name":"Kuaishou Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4782-7261","authenticated-orcid":false,"given":"Xiang","family":"Gao","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5280-5773","authenticated-orcid":false,"given":"YiSha","family":"Li","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-0536-540X","authenticated-orcid":false,"given":"Xin","family":"Li","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-7201-7635","authenticated-orcid":false,"given":"Haiyang","family":"Lu","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1329-3876","authenticated-orcid":false,"given":"Ben","family":"Wang","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,8,3]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543846"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1201\/9781315140223"},{"key":"e_1_3_2_2_3_1","volume-title":"International conference on machine learning. PMLR, 176-185","author":"Anschel Oron","year":"2017","unstructured":"Oron Anschel, Nir Baram, and Nahum Shimkin. 2017. Averaged-dqn: Variance reduction and stabilization for deep reinforcement learning. In International conference on machine learning. PMLR, 176-185."},{"key":"e_1_3_2_2_4_1","unstructured":"Jimmy Lei Ba. 2016. Layer normalization. arXiv preprint arXiv:1607.06450(2016)."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543873.3584640"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2020.106478"},{"key":"e_1_3_2_2_7_1","volume-title":"Soft actor-critic for discrete action settings. arXiv","author":"Christodoulou Petros","year":"2019","unstructured":"Petros Christodoulou. [n.d.]. Soft actor-critic for discrete action settings. arXiv 2019. arXiv preprint arXiv:1910.07207( [n.,d.])."},{"key":"e_1_3_2_2_8_1","volume-title":"Ctrl-z: Recovering from instability in reinforcement learning. arXiv preprint arXiv:1910.03732(2019).","author":"Dasagi Vibhavari","year":"2019","unstructured":"Vibhavari Dasagi, Jake Bruce, Thierry Peynot, and J\u00fcrgen Leitner. 2019. Ctrl-z: Recovering from instability in reinforcement learning. arXiv preprint arXiv:1910.03732(2019)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539597.3570412"},{"key":"e_1_3_2_2_10_1","volume-title":"Sixteenth European Workshop on Reinforcement Learning.","author":"Dohare Shibhansh","year":"2023","unstructured":"Shibhansh Dohare, Qingfeng Lan, and A Rupam Mahmood. 2023. Overcoming policy collapse in deep reinforcement learning. In Sixteenth European Workshop on Reinforcement Learning."},{"key":"e_1_3_2_2_11_1","unstructured":"Vincent Fran\u00e7ois-Lavet Raphael Fonteneau and Damien Ernst. 2015. How to discount deep reinforcement learning: Towards new dynamic strategies. arXiv preprint arXiv:1512.02011(2015)."},{"key":"e_1_3_2_2_12_1","volume-title":"A minimalist approach to offline reinforcement learning. Advances in neural information processing systems","author":"Fujimoto Scott","year":"2021","unstructured":"Scott Fujimoto and Shixiang Shane Gu. 2021. A minimalist approach to offline reinforcement learning. Advances in neural information processing systems, Vol. 34 (2021), 20132-20145."},{"key":"e_1_3_2_2_13_1","volume-title":"International conference on machine learning. PMLR, 1587-1596","author":"Fujimoto Scott","year":"2018","unstructured":"Scott Fujimoto, Herke Hoof, and David Meger. 2018. Addressing function approximation error in actor-critic methods. In International conference on machine learning. PMLR, 1587-1596."},{"key":"e_1_3_2_2_14_1","volume-title":"International conference on machine learning. PMLR","author":"Fujimoto Scott","year":"2019","unstructured":"Scott Fujimoto, David Meger, and Doina Precup. 2019. Off-policy deep reinforcement learning without exploration. In International conference on machine learning. PMLR, 2052-2062."},{"key":"e_1_3_2_2_15_1","article-title":"Variance Reduction Techniques for Gradient Estimates in Reinforcement Learning","volume":"5","author":"Greensmith Evan","year":"2004","unstructured":"Evan Greensmith, Peter L Bartlett, and Jonathan Baxter. 2004. Variance Reduction Techniques for Gradient Estimates in Reinforcement Learning. Journal of Machine Learning Research, Vol. 5, 9 (2004).","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_2_16_1","volume-title":"International conference on machine learning. PMLR","author":"Haarnoja Tuomas","year":"2018","unstructured":"Tuomas Haarnoja, Aurick Zhou, Pieter Abbeel, and Sergey Levine. 2018. Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. In International conference on machine learning. PMLR, 1861-1870."},{"key":"e_1_3_2_2_17_1","volume-title":"The elements of statistical learning: data mining, inference, and prediction","author":"Hastie Trevor","unstructured":"Trevor Hastie, Robert Tibshirani, Jerome H Friedman, and Jerome H Friedman. 2009. The elements of statistical learning: data mining, inference, and prediction. Vol. 2. Springer."},{"key":"e_1_3_2_2_18_1","unstructured":"Geoffrey Hinton. 2015. Distilling the Knowledge in a Neural Network. arXiv preprint arXiv:1503.02531(2015)."},{"key":"e_1_3_2_2_19_1","unstructured":"Eugene Ie Vihan Jain Jing Wang Sanmit Narvekar Ritesh Agarwal Rui Wu Heng-Tze Cheng Tushar Chandra and Craig Boutilier. 2019. SlateQ: A tractable decomposition for reinforcement learning with recommendation sets. (2019)."},{"key":"e_1_3_2_2_20_1","volume-title":"Neural tangent kernel: Convergence and generalization in neural networks. Advances in neural information processing systems","author":"Jacot Arthur","year":"2018","unstructured":"Arthur Jacot, Franck Gabriel, and Cl\u00e9ment Hongler. 2018. Neural tangent kernel: Convergence and generalization in neural networks. Advances in neural information processing systems, Vol. 31 (2018)."},{"key":"e_1_3_2_2_21_1","unstructured":"Eric Jang Shixiang Gu and Ben Poole. 2016. Categorical reparameterization with gumbel-softmax. arXiv preprint arXiv:1611.01144(2016)."},{"key":"e_1_3_2_2_22_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980(2014).","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980(2014)."},{"key":"e_1_3_2_2_23_1","unstructured":"Ilya Kostrikov Ashvin Nair and Sergey Levine. 2021. Offline reinforcement learning with implicit q-learning. arXiv preprint arXiv:2110.06169(2021)."},{"key":"e_1_3_2_2_24_1","volume-title":"Stabilizing off-policy q-learning via bootstrapping error reduction. Advances in neural information processing systems","author":"Kumar Aviral","year":"2019","unstructured":"Aviral Kumar, Justin Fu, Matthew Soh, George Tucker, and Sergey Levine. 2019. Stabilizing off-policy q-learning via bootstrapping error reduction. Advances in neural information processing systems, Vol. 32 (2019)."},{"key":"e_1_3_2_2_25_1","unstructured":"Yuxi Li. 2017. Deep reinforcement learning: An overview. arXiv preprint arXiv:1701.07274(2017)."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3512109"},{"key":"e_1_3_2_2_27_1","unstructured":"TP Lillicrap. 2015. Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971(2015)."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583244"},{"key":"e_1_3_2_2_29_1","volume-title":"Malte Schwarzkopf, and Mohammad Alizadeh.","author":"Mao Hongzi","year":"2018","unstructured":"Hongzi Mao, Shaileshh Bojja Venkatakrishnan, Malte Schwarzkopf, and Mohammad Alizadeh. 2018. Variance reduction for reinforcement learning in input-driven environments. arXiv preprint arXiv:1807.02264(2018)."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3481954"},{"key":"e_1_3_2_2_31_1","unstructured":"Volodymyr Mnih. 2013. Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602(2013)."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"crossref","unstructured":"Volodymyr Mnih Koray Kavukcuoglu David Silver Andrei A Rusu Joel Veness Marc G Bellemare Alex Graves Martin Riedmiller Andreas K Fidjeland Georg Ostrovski et al. 2015. Human-level control through deep reinforcement learning. nature Vol. 518 7540 (2015) 529-533.","DOI":"10.1038\/nature14236"},{"key":"e_1_3_2_2_33_1","volume-title":"Awac: Accelerating online reinforcement learning with offline datasets. arXiv preprint arXiv:2006.09359(2020).","author":"Nair Ashvin","year":"2020","unstructured":"Ashvin Nair, Abhishek Gupta, Murtaza Dalal, and Sergey Levine. 2020. Awac: Accelerating online reinforcement learning with offline datasets. arXiv preprint arXiv:2006.09359(2020)."},{"key":"e_1_3_2_2_34_1","unstructured":"Joshua Romoff Peter Henderson Alexandre Pich\u00e9 Vincent Francois-Lavet and Joelle Pineau. 2018. Reward estimation for variance reduction in deep reinforcement learning. arXiv preprint arXiv:1805.03359(2018)."},{"key":"e_1_3_2_2_35_1","volume-title":"Exploit reward shifting in value-based deep-rl: Optimistic curiosity-based exploration and conservative exploitation via linear reward shaping. Advances in neural information processing systems","author":"Sun Hao","year":"2022","unstructured":"Hao Sun, Lei Han, Rui Yang, Xiaoteng Ma, Jian Guo, and Bolei Zhou. 2022. Exploit reward shifting in value-based deep-rl: Optimistic curiosity-based exploration and conservative exploitation via linear reward shaping. Advances in neural information processing systems, Vol. 35 (2022), 37719-37734."},{"key":"e_1_3_2_2_36_1","volume-title":"Policy gradient methods for reinforcement learning with function approximation. Advances in neural information processing systems","author":"Sutton Richard S","year":"1999","unstructured":"Richard S Sutton, David McAllester, Satinder Singh, and Yishay Mansour. 1999. Policy gradient methods for reinforcement learning with function approximation. Advances in neural information processing systems, Vol. 12 (1999)."},{"key":"e_1_3_2_2_37_1","volume-title":"The self-normalized estimator for counterfactual learning. advances in neural information processing systems","author":"Swaminathan Adith","year":"2015","unstructured":"Adith Swaminathan and Thorsten Joachims. 2015. The self-normalized estimator for counterfactual learning. advances in neural information processing systems, Vol. 28 (2015)."},{"key":"e_1_3_2_2_38_1","unstructured":"Chen Tessler Daniel J Mankowitz and Shie Mannor. 2018. Reward constrained policy optimization. arXiv preprint arXiv:1805.11074(2018)."},{"key":"e_1_3_2_2_39_1","unstructured":"Masatoshi Uehara Chengchun Shi and Nathan Kallus. 2022. A review of off-policy evaluation in reinforcement learning. arXiv preprint arXiv:2212.06355(2022)."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"e_1_3_2_2_41_1","unstructured":"A Vaswani. 2017. Attention is all you need. Advances in Neural Information Processing Systems(2017)."},{"key":"e_1_3_2_2_42_1","unstructured":"Christopher John Cornish Hellaby Watkins. 1989. Learning from delayed rewards. (1989)."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401147"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599473"},{"key":"e_1_3_2_2_45_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Yue Yang","year":"2024","unstructured":"Yang Yue, Rui Lu, Bingyi Kang, Shiji Song, and Gao Huang. 2024. Understanding, predicting and better resolving Q-value divergence in offline-RL. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i8.28783"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539040"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2021.3070203"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219823"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330668"}],"event":{"name":"KDD '25: The 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Toronto ON Canada","acronym":"KDD '25","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3711896.3737264","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T17:58:06Z","timestamp":1777571886000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3711896.3737264"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,3]]},"references-count":50,"alternative-id":["10.1145\/3711896.3737264","10.1145\/3711896"],"URL":"https:\/\/doi.org\/10.1145\/3711896.3737264","relation":{},"subject":[],"published":{"date-parts":[[2025,8,3]]},"assertion":[{"value":"2025-08-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}