{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:34:09Z","timestamp":1750221249806,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":13,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,6,27]],"date-time":"2018-06-27T00:00:00Z","timestamp":1530057600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Youth Innovation Promotion Association of the Chinese Academy of Sciences","award":["20144310, 2016102"],"award-info":[{"award-number":["20144310, 2016102"]}]},{"name":"National Key R&D Program of China","award":["2016QY02D0405"],"award-info":[{"award-number":["2016QY02D0405"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61425016, 61472401, 61722211, 20180290"],"award-info":[{"award-number":["61425016, 61472401, 61722211, 20180290"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"973 Program of China","award":["2014CB340401"],"award-info":[{"award-number":["2014CB340401"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,6,27]]},"DOI":"10.1145\/3209978.3210068","type":"proceedings-article","created":{"date-parts":[[2018,7,2]],"date-time":"2018-07-02T12:12:40Z","timestamp":1530533560000},"page":"885-888","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Reducing Variance in Gradient Bandit Algorithm using Antithetic Variates Method"],"prefix":"10.1145","author":[{"given":"Sihao","family":"Yu","sequence":"first","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Xu","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijng, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanyan","family":"Lan","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiafeng","family":"Guo","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xueqi","family":"Cheng","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2018,6,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.5555\/1622845.1622855"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1093\/biomet\/41.3-4.494"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.5555\/1005332.1044710"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1017\/S0305004100031455"},{"key":"e_1_3_2_1_5_1","unstructured":"Ronald A Howard . 1960. Dynamic programming and markov processes. (1960).  Ronald A Howard . 1960. Dynamic programming and markov processes. (1960)."},{"volume-title":"Proceedings of the 30th International Conference on Machine Learning (ICML-13)","year":"2013","author":"Karnin Zohar","key":"e_1_3_2_1_6_1"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1287\/moor.12.2.262"},{"key":"e_1_3_2_1_8_1","unstructured":"Vijay R Konda and John N Tsitsiklis . 2000. Actor-critic algorithms. In Advances in neural information processing systems. 1008--1014.   Vijay R Konda and John N Tsitsiklis . 2000. Actor-critic algorithms. In Advances in neural information processing systems. 1008--1014."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1080\/01621459.1949.10483310"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.2307\/2342192"},{"volume-title":"Policy Gradient Methods for Reinforcement Learning with Function Approximation. Submitted to Advances in Neural Information Processing Systems","year":"1999","author":"Sutton R. S","key":"e_1_3_2_1_11_1"},{"key":"e_1_3_2_1_12_1","first-page":"1","article-title":"Towards a theory of reinforcement-learning connectionist systems","volume":"4","author":"Williams R. J.","year":"1988","journal-title":"Issues in Education"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"}],"event":{"name":"SIGIR '18: The 41st International ACM SIGIR conference on research and development in Information Retrieval","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"],"location":"Ann Arbor MI USA","acronym":"SIGIR '18"},"container-title":["The 41st International ACM SIGIR Conference on Research &amp; Development in Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3209978.3210068","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3209978.3210068","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T02:07:49Z","timestamp":1750212469000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3209978.3210068"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,6,27]]},"references-count":13,"alternative-id":["10.1145\/3209978.3210068","10.1145\/3209978"],"URL":"https:\/\/doi.org\/10.1145\/3209978.3210068","relation":{},"subject":[],"published":{"date-parts":[[2018,6,27]]},"assertion":[{"value":"2018-06-27","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}