{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T05:10:43Z","timestamp":1784178643158,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,7,19]],"date-time":"2018-07-19T00:00:00Z","timestamp":1531958400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Jiangsu SF","award":["BK20160066"],"award-info":[{"award-number":["BK20160066"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,7,19]]},"DOI":"10.1145\/3219819.3219846","type":"proceedings-article","created":{"date-parts":[[2018,7,19]],"date-time":"2018-07-19T13:05:12Z","timestamp":1532005512000},"page":"368-377","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":113,"title":["Reinforcement Learning to Rank in E-Commerce Search Engine"],"prefix":"10.1145","author":[{"given":"Yujing","family":"Hu","sequence":"first","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qing","family":"Da","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Anxiang","family":"Zeng","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Yu","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yinghui","family":"Xu","sequence":"additional","affiliation":[{"name":"Zhejiang Cainiao Supply Chain Management Co., Ltd., Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2018,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Alizila. 2017. Joe Tsai Looks Beyond Alibaba's RMB 3 Trillion Milestone. http:\/\/www.alizila.com\/joe-tsai-beyond-alibabas-3-trillion-milestone\/.  Alizila. 2017. Joe Tsai Looks Beyond Alibaba's RMB 3 Trillion Milestone. http:\/\/www.alizila.com\/joe-tsai-beyond-alibabas-3-trillion-milestone\/."},{"key":"e_1_3_2_1_2_1","first-page":"397","article-title":"Using Confidence Bounds for Exploitation-Exploration Trade-offs","volume":"3","author":"Auer Peter","year":"2002","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1162\/153244303765208377"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/1102351.1102363"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/1148170.1148205"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/1273496.1273513"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220122"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10791-012-9197-9"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/775047.775067"},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the 20th International Conference on Artificial Intelligence and Statistics (AISTATS)","author":"Katariya Sumeet","year":"2017"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1017984413808"},{"key":"e_1_3_2_1_12_1","volume-title":"Cascading Bandits: Learning to Rank in the Cascade Model Proceedings of the 32nd International Conference on Machine Learning (ICML'15). 767--776.","author":"Kveton Branislav","year":"2015"},{"key":"e_1_3_2_1_13_1","unstructured":"Branislav Kveton Zheng Wen Azin Ashkan and Csaba Szepesvari. 2015. Combinatorial Cascading Bandits. In Advances in Neural Information Processing Systems 28 (NIPS'15). 1450--1458.   Branislav Kveton Zheng Wen Azin Ashkan and Csaba Szepesvari. 2015. Combinatorial Cascading Bandits. In Advances in Neural Information Processing Systems 28 (NIPS'15). 1450--1458."},{"key":"e_1_3_2_1_14_1","unstructured":"Paul Lagr\u00e9e Claire Vernade and Olivier Cappe. 2016. Multiple-Play Bandits in the Position-based Model Advances in Neural Information Processing Systems 29 (NIPS'16). 1597--1605.  Paul Lagr\u00e9e Claire Vernade and Olivier Cappe. 2016. Multiple-Play Bandits in the Position-based Model Advances in Neural Information Processing Systems 29 (NIPS'16). 1597--1605."},{"key":"e_1_3_2_1_15_1","unstructured":"John Langford and Tong Zhang. 2008. The Epoch-greedy Algorithm for Multi-Armed Bandits with Side Information Advances in Neural Information Processing Systems 21 (NIPS'08). 817--824.   John Langford and Tong Zhang. 2008. The Epoch-greedy Algorithm for Multi-Armed Bandits with Side Information Advances in Neural Information Processing Systems 21 (NIPS'08). 817--824."},{"key":"e_1_3_2_1_16_1","volume-title":"Burges","author":"Li Ping","year":"2008"},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the 31st International Conference on Machine Learning (ICML'16)","author":"Li Shuai","year":"2016"},{"key":"e_1_3_2_1_18_1","volume-title":"Continuous Control with Deep Reinforcement Learning. arXiv preprint arXiv:1509.02971","author":"Lillicrap Timothy P.","year":"2015"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1561\/1500000016"},{"key":"e_1_3_2_1_20_1","volume-title":"Toward Off-Policy Learning Control with Function Approximation Proceedings of the 27th International Conference on Machine Learning (ICML'10)","author":"Maei Hamid R."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Volodymyr Mnih Koray Kavukcuoglu David Silver Andrei A. Rusu Joel Veness Marc G. Bellemare Alex Graves Martin Riedmiller Andreas K. Fidjeland Georg Ostrovski etal. 2015. Human-Level Control Through Deep Reinforcement Learning. Nature Vol. 518 7540 (2015) 529--533.  Volodymyr Mnih Koray Kavukcuoglu David Silver Andrei A. Rusu Joel Veness Marc G. Bellemare Alex Graves Martin Riedmiller Andreas K. Fidjeland Georg Ostrovski et al.. 2015. Human-Level Control Through Deep Reinforcement Learning. Nature Vol. 518 7540 (2015) 529--533.","DOI":"10.1038\/nature14236"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/1008992.1009006"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390255"},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of the 32nd International Conference on Machine Learning (ICML'15)","author":"Schulman John","year":"2015"},{"key":"e_1_3_2_1_25_1","volume-title":"Julian Schrittwieser, Ioannis Antonoglou, Veda Panneershelvam, Marc Lanctot, et al..","author":"Silver David","year":"2016"},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of the 31st International Conference on Machine Learning (ICML'14)","author":"Silver David","year":"2014"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.5555\/2567709.2502595"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.5555\/551283"},{"key":"e_1_3_2_1_29_1","unstructured":"Richard S. Sutton David A. McAllester Satinder P. Singh and Yishay Mansour. 2000. Policy Gradient Methods for Reinforcement Learning with Function Approximation Advances in Neural Information Processing Systems 13 (NIPS'00). 1057--1063.   Richard S. Sutton David A. McAllester Satinder P. Singh and Yishay Mansour. 2000. Policy Gradient Methods for Reinforcement Learning with Function Approximation Advances in Neural Information Processing Systems 13 (NIPS'00). 1057--1063."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553527"},{"key":"e_1_3_2_1_33_1","volume-title":"Online Learning to Rank in Stochastic Click Models Proceedings of the 34th International Conference on Machine Learning (ICML'17). 4199--4208","author":"Zoghi Masrour"},{"key":"e_1_3_2_1_34_1","volume-title":"Cascading Bandits for Large-Scale Recommendation Problems Proceedings of the 32nd Conference on Uncertainty in Artificial Intelligence (UAI'16)","author":"Zong Shi","year":"2016"}],"event":{"name":"KDD '18: The 24th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining","location":"London United Kingdom","acronym":"KDD '18","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 24th ACM SIGKDD International Conference on Knowledge Discovery &amp; Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3219819.3219846","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3219819.3219846","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T01:08:43Z","timestamp":1750208923000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3219819.3219846"}},"subtitle":["Formalization, Analysis, and Application"],"short-title":[],"issued":{"date-parts":[[2018,7,19]]},"references-count":33,"alternative-id":["10.1145\/3219819.3219846","10.1145\/3219819"],"URL":"https:\/\/doi.org\/10.1145\/3219819.3219846","relation":{},"subject":[],"published":{"date-parts":[[2018,7,19]]},"assertion":[{"value":"2018-07-19","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}