{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T07:55:16Z","timestamp":1778745316797,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,7,25]],"date-time":"2020-07-25T00:00:00Z","timestamp":1595635200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Natural Science Foundation of China","award":["61872338, 61832017, 61773362"],"award-info":[{"award-number":["61872338, 61832017, 61773362"]}]},{"name":"Beijing Outstanding Young Scientist Program","award":["BJJWZYJH012019100020098"],"award-info":[{"award-number":["BJJWZYJH012019100020098"]}]},{"name":"Youth Innovation Promotion Association CAS","award":["2016102"],"award-info":[{"award-number":["2016102"]}]},{"name":"Fundamental Research Funds for the Central Universities, and Research Funds of Renmin University of China","award":["2018030246"],"award-info":[{"award-number":["2018030246"]}]},{"name":"Beijing Academy of Artificial Intelligence","award":["BAAI2019ZD0305, BAAI2020ZJ0303"],"award-info":[{"award-number":["BAAI2019ZD0305, BAAI2020ZJ0303"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,7,25]]},"DOI":"10.1145\/3397271.3401148","type":"proceedings-article","created":{"date-parts":[[2020,7,25]],"date-time":"2020-07-25T07:50:08Z","timestamp":1595663408000},"page":"509-518","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":25,"title":["Reinforcement Learning to Rank with Pairwise Policy Gradient"],"prefix":"10.1145","author":[{"given":"Jun","family":"Xu","sequence":"first","affiliation":[{"name":"Renmin University of China &amp; Beijing Key Laboratory of Big Data Management and Analysis Methods, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zeng","family":"Wei","sequence":"additional","affiliation":[{"name":"Baidu Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Long","family":"Xia","sequence":"additional","affiliation":[{"name":"York University, Toronto , Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanyan","family":"Lan","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, CAS, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dawei","family":"Yin","sequence":"additional","affiliation":[{"name":"Baidu Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xueqi","family":"Cheng","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, CAS, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ji-Rong","family":"Wen","sequence":"additional","affiliation":[{"name":"Renmin University of China &amp; Beijing Key Laboratory of Big Data Management and Analysis Methods, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2020,7,25]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/1102351.1102363"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/1148170.1148205"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/1273496.1273513"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/290941.291025"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.5555\/1855038"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/1390334.1390446"},{"key":"e_1_3_2_1_8_1","volume-title":"Advances in Neural Information Processing Systems 14","author":"Crammer Koby","unstructured":"Koby Crammer and Yoram Singer . 2002. Pranking with Ranking . In Advances in Neural Information Processing Systems 14 , T. G. Dietterich, S. Becker, and Z. Ghahramani (Eds.). MIT Press , 641--647. Koby Crammer and Yoram Singer. 2002. Pranking with Ranking. In Advances in Neural Information Processing Systems 14, T. G. Dietterich, S. Becker, and Z. Ghahramani (Eds.). MIT Press, 641--647."},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of the 35th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR '12)","author":"Dang Van","unstructured":"Van Dang and W. Bruce Croft . 2012. Diversity by Proportionality: An Election-based Approach to Search Result Diversification . In Proceedings of the 35th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR '12) . ACM, New York, NY, USA, 65--74. Van Dang and W. Bruce Croft. 2012. Diversity by Proportionality: An Election-based Approach to Search Result Diversification. In Proceedings of the 35th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR '12). ACM, New York, NY, USA, 65--74."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3178876.3186165"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3209979"},{"key":"e_1_3_2_1_12_1","volume-title":"Information Retrieval","volume":"16","author":"Hofmann Katja","year":"2013","unstructured":"Katja Hofmann , Shimon Whiteson , and Maarten de Rijke . 2013 a. Balancing exploration and exploitation in listwise and pairwise online learning to rank for information retrieval . Information Retrieval , Vol. 16 , 1 (01 Feb 2013), 63--90. Katja Hofmann, Shimon Whiteson, and Maarten de Rijke. 2013a. Balancing exploration and exploitation in listwise and pairwise online learning to rank for information retrieval. Information Retrieval, Vol. 16, 1 (01 Feb 2013), 63--90."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10791-012-9197-9"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219846"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/582415.582418"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/775047.775067"},{"key":"e_1_3_2_1_17_1","volume-title":"Zheng Wen, and Azin Ashkan.","author":"Kveton Branislav","year":"2015","unstructured":"Branislav Kveton , Csaba Szepesv\u00e1 ri , Zheng Wen, and Azin Ashkan. 2015 . Cascading Bandits : Learning to Rank in the Cascade Model. CoRR , Vol. abs\/ 1502 .02763 (2015). Branislav Kveton, Csaba Szepesv\u00e1 ri, Zheng Wen, and Azin Ashkan. 2015. Cascading Bandits: Learning to Rank in the Cascade Model. CoRR, Vol. abs\/1502.02763 (2015)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.2200\/S00607ED2V01Y201410HLT026"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/2911451.2911548"},{"key":"e_1_3_2_1_20_1","first-page":"3","article-title":"Learning to Rank for Information Retrieval","volume":"3","author":"Liu Tie-Yan","year":"2009","unstructured":"Tie-Yan Liu . 2009 . Learning to Rank for Information Retrieval . Found. Trends Inf. Retr. , Vol. 3 , 3 (March 2009), 225--331. Tie-Yan Liu. 2009. Learning to Rank for Information Retrieval. Found. Trends Inf. Retr., Vol. 3, 3 (March 2009), 225--331.","journal-title":"Found. Trends Inf. Retr."},{"key":"e_1_3_2_1_21_1","volume-title":"Partially Observable Markov Decision Process for Recommender Systems. CoRR","author":"Lu Zhongqi","year":"2016","unstructured":"Zhongqi Lu and Qiang Yang . 2016. Partially Observable Markov Decision Process for Recommender Systems. CoRR , Vol. abs\/ 1608 .07793 ( 2016 ). Zhongqi Lu and Qiang Yang. 2016. Partially Observable Markov Decision Process for Recommender Systems. CoRR, Vol. abs\/1608.07793 (2016)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/2600428.2609629"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/1008992.1009006"},{"key":"e_1_3_2_1_24_1","volume-title":"Ranking for Relevance and Display Preferences in Complex Presentation Layouts. In The 41st International ACM SIGIR Conference on Research & Development in Information Retrieval (SIGIR '18)","author":"Oosterhuis Harrie","year":"2018","unstructured":"Harrie Oosterhuis and Maarten de Rijke . 2018 . Ranking for Relevance and Display Preferences in Complex Presentation Layouts. In The 41st International ACM SIGIR Conference on Research & Development in Information Retrieval (SIGIR '18) . 845--854. Harrie Oosterhuis and Maarten de Rijke. 2018. Ranking for Relevance and Display Preferences in Complex Presentation Layouts. In The 41st International ACM SIGIR Conference on Research & Development in Information Retrieval (SIGIR '18). 845--854."},{"key":"e_1_3_2_1_25_1","first-page":"4","article-title":"LETOR: A Benchmark Collection for Research on Learning to Rank for Information","volume":"13","author":"Qin Tao","year":"2010","unstructured":"Tao Qin , Tie-Yan Liu , Jun Xu , and Hang Li . 2010 . LETOR: A Benchmark Collection for Research on Learning to Rank for Information Retrieval. Inf. Retr. , Vol. 13 , 4 (Aug. 2010), 346--374. Tao Qin, Tie-Yan Liu, Jun Xu, and Hang Li. 2010. LETOR: A Benchmark Collection for Research on Learning to Rank for Information Retrieval. Inf. Retr., Vol. 13, 4 (Aug. 2010), 346--374.","journal-title":"Retrieval. Inf. Retr."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390255"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390255"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/1772690.1772780"},{"key":"e_1_3_2_1_29_1","volume-title":"Brafman","author":"Shani Guy","year":"2005","unstructured":"Guy Shani , David Heckerman , and Ronen I . Brafman . 2005 . An MDP-Based Recommender System. J. Mach. Learn. Res ., Vol. 6 (Dec. 2005), 1265--1295. Guy Shani, David Heckerman, and Ronen I. Brafman. 2005. An MDP-Based Recommender System. J. Mach. Learn. Res., Vol. 6 (Dec. 2005), 1265--1295."},{"key":"e_1_3_2_1_30_1","volume-title":"Virtual-Taobao: Virtualizing Real-world Online Retail Environment for Reinforcement Learning. CoRR","author":"Shi Jing-Cheng","year":"2018","unstructured":"Jing-Cheng Shi , Yang Yu , Qing Da , Shi-Yong Chen , and Anxiang Zeng . 2018. Virtual-Taobao: Virtualizing Real-world Online Retail Environment for Reinforcement Learning. CoRR , Vol. abs\/ 1805 .10000 ( 2018 ). Jing-Cheng Shi, Yang Yu, Qing Da, Shi-Yong Chen, and Anxiang Zeng. 2018. Virtual-Taobao: Virtualizing Real-world Online Retail Environment for Reinforcement Learning. CoRR, Vol. abs\/1805.10000 (2018)."},{"key":"e_1_3_2_1_31_1","volume-title":"Barto","author":"Sutton Richard S.","year":"2016","unstructured":"Richard S. Sutton and Andrew G . Barto . 2016 . Reinforcement Learning : An Introduction 2nd ed.). MIT Press . Richard S. Sutton and Andrew G. Barto. 2016. Reinforcement Learning: An Introduction 2nd ed.). MIT Press."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080786"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2766462.2767710"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/2911451.2911498"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080775"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/1277741.1277809"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/2747874"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390310"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553527"},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of the 40th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR '17)","author":"Zeng Wei","year":"2017","unstructured":"Wei Zeng , Jun Xu , Yanyan Lan , Jiafeng Guo , and Xueqi Cheng . 2017 . Reinforcement Learning to Rank with Markov Decision Process . In Proceedings of the 40th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR '17) . 945--948. Wei Zeng, Jun Xu, Yanyan Lan, Jiafeng Guo, and Xueqi Cheng. 2017. Reinforcement Learning to Rank with Markov Decision Process. In Proceedings of the 40th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR '17). 945--948."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3234944.3234977"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/2600428.2609529"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240323.3240374"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219886"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/2600428.2609634"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330668"}],"event":{"name":"SIGIR '20: The 43rd International ACM SIGIR conference on research and development in Information Retrieval","location":"Virtual Event China","acronym":"SIGIR '20","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 43rd International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3397271.3401148","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3397271.3401148","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T22:41:43Z","timestamp":1750200103000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3397271.3401148"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,7,25]]},"references-count":45,"alternative-id":["10.1145\/3397271.3401148","10.1145\/3397271"],"URL":"https:\/\/doi.org\/10.1145\/3397271.3401148","relation":{},"subject":[],"published":{"date-parts":[[2020,7,25]]},"assertion":[{"value":"2020-07-25","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}