{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:30:10Z","timestamp":1750221010427,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,7,25]],"date-time":"2019-07-25T00:00:00Z","timestamp":1564012800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,7,25]]},"DOI":"10.1145\/3292500.3330864","type":"proceedings-article","created":{"date-parts":[[2019,7,26]],"date-time":"2019-07-26T13:17:26Z","timestamp":1564147046000},"page":"1184-1193","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Off-policy Learning for Multiple Loggers"],"prefix":"10.1145","author":[{"given":"Li","family":"He","sequence":"first","affiliation":[{"name":"JD.com, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Long","family":"Xia","sequence":"additional","affiliation":[{"name":"JD.com, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Zeng","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, CAS, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhi-Ming","family":"Ma","sequence":"additional","affiliation":[{"name":"Academy of Mathematics and Systems Science, CAS, Beijing, Chile"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yihong","family":"Zhao","sequence":"additional","affiliation":[{"name":"JD.com, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dawei","family":"Yin","sequence":"additional","affiliation":[{"name":"JD.com, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2019,7,25]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3097983.3098155"},{"volume-title":"Schapire","year":"2014","author":"Agarwal Alekh","key":"e_1_3_2_1_2_1"},{"volume-title":"Efficient Policy Learning. CoRR","year":"2017","author":"Athey Susan","key":"e_1_3_2_1_3_1"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1080\/01621459.1962.10482149"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10107-017-1125-8"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/1557019.1557040"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.5555\/2567709.2567766"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1093\/biomet\/63.3.615"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/1961189.1961199"},{"volume-title":"Learning Bounds for Importance Weighting. In NIPS'10","year":"2010","author":"Cortes Corinna","key":"e_1_3_2_1_10_1"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1561\/0100000004"},{"volume-title":"Doubly Robust Policy Evaluation and Learning. In ICML'11","year":"2011","author":"Miroslav Dud'i","key":"e_1_3_2_1_12_1"},{"volume-title":"More Robust Doubly Robust Off-policy Evaluation. In ICML'18","year":"2018","author":"Farajtabar Mehrdad","key":"e_1_3_2_1_13_1"},{"volume-title":"Fundamentals of convex analysis","author":"Hiriart-Urruty Lemar\u00e9chal Claude","key":"e_1_3_2_1_14_1"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(89)90020-8"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1080\/01621459.1952.10483446"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1198\/106186008X320456"},{"volume-title":"Categorical Reparameterization with Gumbel-Softmax. In ICLR'16 .","year":"2016","author":"Jang Eric","key":"e_1_3_2_1_18_1"},{"volume-title":"Doubly Robust Off-policy Value Evaluation for Reinforcement Learning. In ICML'16","year":"2016","author":"Jiang Nan","key":"e_1_3_2_1_19_1"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/2911451.2914803"},{"volume-title":"Unbiased Learning-to-Rank with Biased Feedback. In IJCAI'18","year":"2018","author":"Joachims Thorsten","key":"e_1_3_2_1_21_1"},{"volume-title":"Kingma and Jimmy Ba","year":"2015","author":"Diederik","key":"e_1_3_2_1_22_1"},{"volume-title":"Pereira","year":"2001","author":"Lafferty John D.","key":"e_1_3_2_1_23_1"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10107-012-0514-2"},{"volume-title":"Counterfactual Estimation and Optimization of Click Metrics for Search Engines. CoRR","year":"1891","author":"Li Lihong","key":"e_1_3_2_1_25_1"},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of the 2011 International Conference on On-line Trading of Exploration and Exploitation 2 -","volume":"26","author":"Li Lihong","year":"2011"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/1935826.1935878"},{"volume-title":"The Concrete Distribution: A Continuous Relaxation of Discrete Random Variables. CoRR","year":"2016","author":"Maddison Chris J.","key":"e_1_3_2_1_28_1"},{"volume-title":"Safe and Efficient Off-Policy Reinforcement Learning. In NIPS'16","author":"Munos R\u00e9","key":"e_1_3_2_1_29_1"},{"volume-title":"Variance-based Regularization with Convex Objectives. In NIPS'17","author":"Namkoong Hongseok","key":"e_1_3_2_1_30_1"},{"volume-title":"Efficient Counterfactual Learning from Bandit Feedback. CoRR","year":"2018","author":"Narita Yusuke","key":"e_1_3_2_1_31_1"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2010.2068870"},{"volume-title":"NIPS'16","year":"2016","author":"Nowozin Sebastian","key":"e_1_3_2_1_33_1"},{"key":"e_1_3_2_1_34_1","unstructured":"Art B. Owen. 2013. Monte Carlo theory methods and examples .  Art B. Owen. 2013. Monte Carlo theory methods and examples ."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.5555\/1953048.2078195"},{"volume-title":"Eligibility Traces for Off-Policy Policy Evaluation. In ICML'00","author":"Precup Doina","key":"e_1_3_2_1_36_1"},{"key":"e_1_3_2_1_37_1","volume-title":"Proceedings of the Fourth Berkeley Symposium on Mathematical Statistics and Probability","volume":"561","author":"Alfr\u00e9","year":"1961"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1093\/biomet\/70.1.41"},{"volume-title":"ICML'17","year":"2017","author":"Shalit Uri","key":"e_1_3_2_1_39_1"},{"volume-title":"Multi-armed Bandit Problems with History. In AISTATS'12","author":"Pannagadatta","key":"e_1_3_2_1_40_1"},{"volume-title":"NIPS'10","year":"2010","author":"Strehl Alexander L.","key":"e_1_3_2_1_41_1"},{"volume-title":"ICML Workshop on Machine Learning for Causal Inference, Counterfactual Prediction, and Autonomous Action (CausalML) .","author":"Su Yi","key":"e_1_3_2_1_42_1"},{"volume-title":"CAB: Continuous Adaptive Blending for Policy Evaluation and Learning. In ICML'19 .","year":"2019","author":"Su Yi","key":"e_1_3_2_1_43_1"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"volume-title":"ICML'15","year":"2015","author":"Swaminathan Adith","key":"e_1_3_2_1_45_1"},{"volume-title":"The Self-Normalized Estimator for Counterfactual Learning. In NIPS'15","year":"2015","author":"Swaminathan Adith","key":"e_1_3_2_1_46_1"},{"volume-title":"NIPS'17","year":"2017","author":"Swaminathan Adith","key":"e_1_3_2_1_47_1"},{"volume-title":"Data-Efficient Off-Policy Policy Evaluation for Reinforcement Learning. In ICML'16","author":"Philip","key":"e_1_3_2_1_48_1"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3159652.3159732"},{"volume-title":"ICML'18","author":"Wu Hang","key":"e_1_3_2_1_50_1"},{"volume-title":"Deep Reinforcement Learning for Search, Recommendation, and Online Advertising: A Survey. CoRR","year":"2018","author":"Zhao Xiangyu","key":"e_1_3_2_1_51_1"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240323.3240374"},{"volume-title":"Model-Based Reinforcement Learning for Whole-Chain Recommendations. CoRR","year":"2019","author":"Zhao Xiangyu","key":"e_1_3_2_1_53_1"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219886"},{"volume-title":"2018 d. Deep Reinforcement Learning for List-wise Recommendations. CoRR","year":"2018","author":"Zhao Xiangyu","key":"e_1_3_2_1_55_1"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330668"}],"event":{"name":"KDD '19: The 25th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"],"location":"Anchorage AK USA","acronym":"KDD '19"},"container-title":["Proceedings of the 25th ACM SIGKDD International Conference on Knowledge Discovery &amp; Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3292500.3330864","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3292500.3330864","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T00:26:02Z","timestamp":1750206362000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3292500.3330864"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,7,25]]},"references-count":56,"alternative-id":["10.1145\/3292500.3330864","10.1145\/3292500"],"URL":"https:\/\/doi.org\/10.1145\/3292500.3330864","relation":{},"subject":[],"published":{"date-parts":[[2019,7,25]]},"assertion":[{"value":"2019-07-25","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}