{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:08:44Z","timestamp":1784138924511,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3809577","type":"proceedings-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T14:28:19Z","timestamp":1783693699000},"page":"1823-1833","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Large-Scale Online Learning for Generative List Recommendation in E-commerce: An Environment Policy Optimization Approach"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3630-9179","authenticated-orcid":false,"given":"Yuan","family":"Wang","sequence":"first","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8187-4177","authenticated-orcid":false,"given":"Zhiyu","family":"Li","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8964-801X","authenticated-orcid":false,"given":"Ang","family":"Gao","sequence":"additional","affiliation":[{"name":"Gaoling School of Artificial Intelligence, Renmin University of China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8481-9421","authenticated-orcid":false,"given":"Changshuo","family":"Zhang","sequence":"additional","affiliation":[{"name":"Gaoling School of Artificial Intelligence, Renmin University of China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7397-5632","authenticated-orcid":false,"given":"Xiao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Gaoling School of Artificial Intelligence, Renmin University of China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7170-111X","authenticated-orcid":false,"given":"Jun","family":"Xu","sequence":"additional","affiliation":[{"name":"Gaoling School of Artificial Intelligence, Renmin University of China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7149-5525","authenticated-orcid":false,"given":"Quan","family":"Lin","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3209985"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341981.3344218"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511546921"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3038912.3052627"},{"key":"e_1_3_2_1_5_1","volume-title":"Advances in Neural Information Processing Systems","volume":"24","author":"Chapelle Olivier","year":"2011","unstructured":"Olivier Chapelle and Lihong Li. 2011. An empirical evaluation of Thompson sampling. In Advances in Neural Information Processing Systems, Vol. 24."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599796"},{"key":"e_1_3_2_1_7_1","volume-title":"GRN: Generative Rerank Network for Context-wise Recommendation. arXiv preprint arXiv:2104.00860","author":"Feng Yufei","year":"2021","unstructured":"Yufei Feng, Binbin Hu, Yu Gong, Fei Sun, Qingwen Liu, and Wenwu Ou. 2021. GRN: Generative Rerank Network for Context-wise Recommendation. arXiv preprint arXiv:2104.00860 (2021)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330832"},{"key":"e_1_3_2_1_9_1","volume-title":"Stochastic optimization of sorting networks via continuous relaxations. arXiv preprint arXiv:1903.08850","author":"Grover Aditya","year":"2019","unstructured":"Aditya Grover, Eric Wang, Aaron Zweig, and Stefano Ermon. 2019. Stochastic optimization of sorting networks via continuous relaxations. arXiv preprint arXiv:1903.08850 (2019)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467086"},{"key":"e_1_3_2_1_11_1","volume-title":"International conference on machine learning. PMLR","author":"Haarnoja Tuomas","year":"2018","unstructured":"Tuomas Haarnoja, Aurick Zhou, Pieter Abbeel, and Sergey Levine. 2018. Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. In International conference on machine learning. PMLR, 1861-1870."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1561\/2400000013"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Guangda Huzhang Zhenjia Pang Yongqing Gao Yawen Liu Weijie Shen Wen-Ji Zhou Qing Da Anxiang Zeng Han Yu Yang Yu et al. 2021. AliExpress Learning-To-Rank: Maximizing online model performance without going online. IEEE Transactions on Knowledge and Data Engineering (2021).","DOI":"10.1109\/TKDE.2021.3098898"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080805"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.5555\/1622737.1622748"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3213586.3226206"},{"key":"e_1_3_2_1_17_1","volume-title":"Actor-critic algorithms. Advances in neural information processing systems","author":"Konda Vijay","year":"1999","unstructured":"Vijay Konda and John Tsitsiklis. 1999. Actor-critic algorithms. Advances in neural information processing systems, Vol. 12 (1999)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2016.08.014"},{"key":"e_1_3_2_1_19_1","volume-title":"Deep reinforcement learning: An overview. arXiv preprint arXiv:1701.07274","author":"Yuxi Li.","year":"2017","unstructured":"Yuxi Li. 2017. Deep reinforcement learning: An overview. arXiv preprint arXiv:1701.07274 (2017)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401084"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539618.3591747"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599364"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442381.3449864"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2025\/510"},{"key":"e_1_3_2_1_25_1","volume-title":"Asynchronous Methods for Deep Reinforcement Learning. arXiv preprint arXiv:1602.01783","author":"Mnih Volodymyr","year":"2016","unstructured":"Volodymyr Mnih. 2016. Asynchronous Methods for Deep Reinforcement Learning. arXiv preprint arXiv:1602.01783 (2016)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Volodymyr Mnih Koray Kavukcuoglu David Silver Andrei A Rusu Joel Veness Marc G Bellemare Alex Graves Martin Riedmiller Andreas K Fidjeland Georg Ostrovski et al. 2015. Human-level control through deep reinforcement learning. nature Vol. 518 7540 (2015) 529-533.","DOI":"10.1038\/nature14236"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462830"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401104"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3298689.3347000"},{"key":"e_1_3_2_1_30_1","volume-title":"Trust Region Policy Optimization. arXiv preprint arXiv:1502.05477","author":"Schulman John","year":"2015","unstructured":"John Schulman. 2015. Trust Region Policy Optimization. arXiv preprint arXiv:1502.05477 (2015)."},{"key":"e_1_3_2_1_31_1","volume-title":"Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347","author":"Schulman John","year":"2017","unstructured":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov. 2017. Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1561\/2200000018"},{"key":"e_1_3_2_1_33_1","unstructured":"Zhihong Shao Peiyi Wang Qihao Zhu Runxin Xu Junxiao Song Xiao Bi Haowei Zhang Mingchuan Zhang Y. K. Li Y. Wu and Daya Guo. 2024. DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models. arXiv:2402.03300 [cs.CL]"},{"key":"e_1_3_2_1_34_1","volume-title":"2026 a. Enhancing Bandit Algorithms with LLMs for Time-varying User Preferences in Streaming Recommendations. ACM Transactions on Information Systems","author":"Shen Chenglei","year":"2026","unstructured":"Chenglei Shen, Yi Zhan, Weijie Yu, Xiao Zhang, and Jun Xu. 2026 a. Enhancing Bandit Algorithms with LLMs for Time-varying User Preferences in Streaming Recommendations. ACM Transactions on Information Systems (2026)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11704-025-41366-5"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3614921"},{"key":"e_1_3_2_1_37_1","volume-title":"Attention is all you need. Advances in neural information processing systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_38_1","volume-title":"Sequential evaluation and generation framework for combinatorial recommender system. arXiv preprint arXiv:1902.00245","author":"Wang Fan","year":"2019","unstructured":"Fan Wang, Xiaomin Fang, Lihang Liu, Yaxue Chen, Jiucheng Tao, Zhiming Peng, Cihang Jin, and Hao Tian. 2019. Sequential evaluation and generation framework for combinatorial recommender system. arXiv preprint arXiv:1902.00245 (2019)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3640457.3688165"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599536"},{"key":"e_1_3_2_1_41_1","volume-title":"Reinforcement learning. Adaptation, learning, and optimization","author":"Wiering Marco A","year":"2012","unstructured":"Marco A Wiering and Martijn Van Otterlo. 2012. Reinforcement learning. Adaptation, learning, and optimization, Vol. 12, 3 (2012), 729."},{"key":"e_1_3_2_1_42_1","volume-title":"Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine learning","author":"Williams Ronald J","year":"1992","unstructured":"Ronald J Williams. 1992. Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine learning, Vol. 8, 3 (1992), 229-256."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/2911451.2911498"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080775"},{"key":"e_1_3_2_1_45_1","unstructured":"Fengli Xu Qianyue Hao Zefang Zong Jingwei Wang Yunke Zhang Jingyi Wang Xiaochong Lan Jiahui Gong Tianjian Ouyang Fanjin Meng Chenyang Shao Yuwei Yan Qinglong Yang Yiwen Song Sijian Ren Xinyuan Hu Yu Li Jie Feng Chen Gao and Yong Li. 2025. Towards Large Reasoning Models: A Survey of Reinforced Reasoning with Large Language Models. arXiv:2501.09686 [cs.AI]"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i5.16587"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657714"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3705328.3748060"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539393"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i12.17305"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1002\/widm.1295"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2818"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462948"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:27:36Z","timestamp":1784136456000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3809577"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":53,"alternative-id":["10.1145\/3805712.3809577","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3809577","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}