{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,4]],"date-time":"2026-04-04T06:14:50Z","timestamp":1775283290737,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":41,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,11,3]],"date-time":"2019-11-03T00:00:00Z","timestamp":1572739200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Natural Science Foundation of China","award":["61622208, 61732008, 61532011"],"award-info":[{"award-number":["61622208, 61732008, 61532011"]}]},{"name":"National Key Research and Development Program of China","award":["2018YFC0831700"],"award-info":[{"award-number":["2018YFC0831700"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,11,3]]},"DOI":"10.1145\/3357384.3357945","type":"proceedings-article","created":{"date-parts":[[2019,11,4]],"date-time":"2019-11-04T14:11:35Z","timestamp":1572876695000},"page":"1603-1612","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["Context-Aware Ranking by Constructing a Virtual Environment for Reinforcement Learning"],"prefix":"10.1145","author":[{"given":"Junqi","family":"Zhang","sequence":"first","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiaxin","family":"Mao","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yiqun","family":"Liu","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruizhe","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Min","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shaoping","family":"Ma","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Xu","sequence":"additional","affiliation":[{"name":"Renmin University of China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qi","family":"Tian","sequence":"additional","affiliation":[{"name":"Huawei Noah's Ark Lab, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2019,11,3]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Qingyao Ai Keping Bi Luo Cheng Jiafeng Guo and W. Bruce Croft. 2018. Unbiased Learning to Rank with Unbiased Propensity Estimation. (2018).  Qingyao Ai Keping Bi Luo Cheng Jiafeng Guo and W. Bruce Croft. 2018. Unbiased Learning to Rank with Unbiased Propensity Estimation. (2018)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/2872427.2883033"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Alexey Borisov Martijn Wardenaar Ilya Markov and Maarten De Rijke. 2018. A Click Sequence Model for Web Search. (2018).  Alexey Borisov Martijn Wardenaar Ilya Markov and Maarten De Rijke. 2018. A Click Sequence Model for Web Search. (2018).","DOI":"10.1145\/3209978.3210004"},{"key":"e_1_3_2_1_4_1","volume-title":"Active Object Localization with Deep Reinforcement Learning. In IEEE International Conference on Computer Vision .","author":"Juan"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Jaime Carbonell and Jade Goldstein. 1999. The Use of MMR Diversity-Based Reranking for Reordering Documents and Producing Summaries .  Jaime Carbonell and Jade Goldstein. 1999. The Use of MMR Diversity-Based Reranking for Reordering Documents and Producing Summaries .","DOI":"10.1145\/290941.291025"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/1526709.1526711"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/2124295.2124351"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Ye Chen Yiqun Liu Ke Zhou Meng Wang Min Zhang and Shaoping Ma. 2015. Does Vertical Bring more Satisfaction?:Predicting Search Satisfaction in a Heterogeneous Environment. (2015) 1581--1590.  Ye Chen Yiqun Liu Ke Zhou Meng Wang Min Zhang and Shaoping Ma. 2015. Does Vertical Bring more Satisfaction?:Predicting Search Satisfaction in a Heterogeneous Environment. (2015) 1581--1590.","DOI":"10.1145\/2806416.2806473"},{"key":"e_1_3_2_1_9_1","volume-title":"Meta-evaluation of Online and Offline Web Search Evaluation Metrics. In International Acm Sigir Conference .","author":"Chen Ye","year":"2017"},{"key":"e_1_3_2_1_10_1","volume-title":"Dzmitry Bahdanau, and Yoshua Bengio.","author":"Cho Kyunghyun","year":"2014"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-36973-5_1"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/1390334.1390392"},{"key":"e_1_3_2_1_13_1","volume-title":"Eyetracking in Online Search. Bericht","author":"Granka Laura","year":"2008"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/1240624.1240691"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/1498759.1498818"},{"key":"e_1_3_2_1_16_1","unstructured":"Shengbo Guo and Scott Sanner. 2010. Probabilistic latent maximal marginal relevance. (2010).  Shengbo Guo and Scott Sanner. 2010. Probabilistic latent maximal marginal relevance. (2010)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Thorsten Joachims Laura Granka Bing Pan Helene Hembrooke and Geri Gay. 2005. Accurately Interpreting Clickthrough Data as Implicit Feedback. 4--11.  Thorsten Joachims Laura Granka Bing Pan Helene Hembrooke and Geri Gay. 2005. Accurately Interpreting Clickthrough Data as Implicit Feedback. 4--11.","DOI":"10.1145\/3130332.3130334"},{"key":"e_1_3_2_1_18_1","volume-title":"Moore","author":"Kaelbling Leslie Pack","year":"1998"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/11871842_29"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3329188"},{"key":"e_1_3_2_1_21_1","volume-title":"Constructing Click Models for Mobile Search. In International Acm Sigir Conference .","author":"Mao Jiaxin","year":"2018"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/1416950.1416952"},{"key":"e_1_3_2_1_23_1","unstructured":"Seyed Sajad Mousavi Michael Schukat and Enda Howley. 2018. Deep Reinforcement Learning: An Overview. (2018).  Seyed Sajad Mousavi Michael Schukat and Enda Howley. 2018. Deep Reinforcement Learning: An Overview. (2018)."},{"key":"e_1_3_2_1_24_1","volume-title":"Algorithms for Inverse Reinforcement Learning. In International Conference on Machine Learning .","author":"Andrew"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Harrie Oosterhuis and Maarten De Rijke. 2018. Ranking for Relevance and Display Preferences in Complex Presentation Layouts. (2018) 845--854.  Harrie Oosterhuis and Maarten De Rijke. 2018. Ranking for Relevance and Display Preferences in Complex Presentation Layouts. (2018) 845--854.","DOI":"10.1145\/3209978.3209992"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10791-009-9123-y"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/582415.582418"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10584-0_3"},{"key":"e_1_3_2_1_29_1","unstructured":"Jing-Cheng Shi Yang Yu Qing Da Shi-Yong Chen and An-Xiang Zeng. 2018. Virtual-taobao: virtualizing real-world online retail environment for reinforcement learning. (2018) 845--854.  Jing-Cheng Shi Yang Yu Qing Da Shi-Yong Chen and An-Xiang Zeng. 2018. Virtual-taobao: virtualizing real-world online retail environment for reinforcement learning. (2018) 845--854."},{"key":"e_1_3_2_1_30_1","volume-title":"Nature","volume":"529","author":"Silver Huang A.","year":"2016"},{"key":"e_1_3_2_1_31_1","volume-title":"Tsung Hsien Wen, and Steve Young","author":"Su Pei Hao","year":"2016"},{"key":"e_1_3_2_1_32_1","unstructured":"Richard S Sutton and Andrew G Barto. 2011. Reinforcement learning: An introduction. (2011).  Richard S Sutton and Andrew G Barto. 2011. Reinforcement learning: An introduction. (2011)."},{"key":"e_1_3_2_1_33_1","volume-title":"Nature","volume":"518","author":"Volodymyr Mnih","year":"2015"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/2484028.2484036"},{"key":"e_1_3_2_1_35_1","first-page":"19","article-title":"Optimizing Whole-Page Presentation for Web Search","volume":"12","author":"Wang Yue","year":"2018","journal-title":"ACM Transactions on the Web (TWEB)"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080685"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"Long Xia Jun Xu Yanyan Lan Jiafeng Guo and Xueqi Cheng. 2015. Learning Maximal Marginal Relevance Model via Directly Optimizing Diversity Evaluation Measures. (2015).  Long Xia Jun Xu Yanyan Lan Jiafeng Guo and Xueqi Cheng. 2015. Learning Maximal Marginal Relevance Model via Directly Optimizing Diversity Evaluation Measures. (2015).","DOI":"10.1145\/2766462.2767710"},{"key":"e_1_3_2_1_38_1","volume-title":"Adapting Markov Decision Process for Search Result Diversification. In International Acm Sigir Conference on Research & Development in Information Retrieval .","author":"Xia Long","year":"2017"},{"key":"e_1_3_2_1_39_1","volume-title":"Towards Vision-Based Deep Reinforcement Learning for Robotic Motion Control. Computer Science","author":"Zhang Fangyi","year":"2015"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3269206.3271673"},{"key":"e_1_3_2_1_41_1","volume-title":"Learning for Search Result Diversification. In International Acm Sigir Conference on Research & Development in Information Retrieval .","author":"Zhu Yadong","year":"2014"}],"event":{"name":"CIKM '19: The 28th ACM International Conference on Information and Knowledge Management","location":"Beijing China","acronym":"CIKM '19","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 28th ACM International Conference on Information and Knowledge Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3357384.3357945","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3357384.3357945","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:23:04Z","timestamp":1750202584000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3357384.3357945"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,11,3]]},"references-count":41,"alternative-id":["10.1145\/3357384.3357945","10.1145\/3357384"],"URL":"https:\/\/doi.org\/10.1145\/3357384.3357945","relation":{},"subject":[],"published":{"date-parts":[[2019,11,3]]},"assertion":[{"value":"2019-11-03","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}