{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T15:51:02Z","timestamp":1774453862065,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,8,14]],"date-time":"2022-08-14T00:00:00Z","timestamp":1660435200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,8,14]]},"DOI":"10.1145\/3534678.3539211","type":"proceedings-article","created":{"date-parts":[[2022,8,12]],"date-time":"2022-08-12T19:06:41Z","timestamp":1660331201000},"page":"4021-4031","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["ROI-Constrained Bidding via Curriculum-Guided Bayesian Reinforcement Learning"],"prefix":"10.1145","author":[{"given":"Haozhe","family":"Wang","sequence":"first","affiliation":[{"name":"ShanghaiTech University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chao","family":"Du","sequence":"additional","affiliation":[{"name":"Alibaba Group, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Panyan","family":"Fang","sequence":"additional","affiliation":[{"name":"Alibaba Group, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuo","family":"Yuan","sequence":"additional","affiliation":[{"name":"Alibaba Group, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuming","family":"He","sequence":"additional","affiliation":[{"name":"ShanghaiTech University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liang","family":"Wang","sequence":"additional","affiliation":[{"name":"Alibaba Group, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Zheng","sequence":"additional","affiliation":[{"name":"Alibaba Group, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,8,14]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"N. Abramson D. J. Braverman and G. S. Sebestyen. 2006. Pattern Recognition and Machine Learning. Publications of the American Statistical Association (2006)."},{"key":"e_1_3_2_2_2_1","volume-title":"International conference on machine learning. PMLR.","author":"Achiam Joshua","year":"2017","unstructured":"Joshua Achiam, David Held, Aviv Tamar, and Pieter Abbeel. 2017. Constrained policy optimization. In International conference on machine learning. PMLR."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"S. Balseiro A. Kim M. Mahdian and V. Mirrokni. 2021. Budget-Management Strategies in Repeated Auctions. Operations Research 69 3 (2021).","DOI":"10.1287\/opre.2020.2073"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553380"},{"key":"e_1_3_2_2_5_1","volume-title":"Variational inference: A review for statisticians. Journal of the American statistical Association","author":"Blei David M","year":"2017","unstructured":"David M Blei, Alp Kucukelbir, and Jon D McAuliffe. 2017. Variational inference: A review for statisticians. Journal of the American statistical Association (2017)."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3018661.3018702"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.5555\/3122009.3242024"},{"key":"e_1_3_2_2_8_1","volume-title":"Lyapunov-based safe policy optimization for continuous control. arXiv preprint arXiv:1901.10031","author":"Chow Yinlam","year":"2019","unstructured":"Yinlam Chow, Ofir Nachum, Aleksandra Faust, Edgar Duenez-Guzman, and Mohammad Ghavamzadeh. 2019. Lyapunov-based safe policy optimization for continuous control. arXiv preprint arXiv:1901.10031 (2019)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/2959100.2959190"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467089"},{"key":"e_1_3_2_2_12_1","volume-title":"Internet advertising and the generalized second-price auction: Selling billions of dollars worth of keywords. American economic review","author":"Edelman Benjamin","year":"2007","unstructured":"Benjamin Edelman, Michael Ostrovsky, and Michael Schwarz. 2007. Internet advertising and the generalized second-price auction: Selling billions of dollars worth of keywords. American economic review (2007)."},{"key":"e_1_3_2_2_13_1","volume-title":"Bayesian Reinforcement Learning: A Survey. CoRR abs\/1609.04436","author":"Ghavamzadeh Mohammad","year":"2016","unstructured":"Mohammad Ghavamzadeh, Shie Mannor, Joelle Pineau, and Aviv Tamar. 2016. Bayesian Reinforcement Learning: A Survey. CoRR abs\/1609.04436 (2016)."},{"key":"e_1_3_2_2_14_1","volume-title":"Jason Cheuk Nam Liang, and Vahab Mirrokni","author":"Golrezaei Negin","year":"2021","unstructured":"Negin Golrezaei, Patrick Jaillet, Jason Cheuk Nam Liang, and Vahab Mirrokni. 2021. Bidding and Pricing in Budget and ROI Constrained Markets. (2021)."},{"key":"e_1_3_2_2_15_1","volume-title":"International conference on machine learning. PMLR.","author":"Haarnoja Tuomas","year":"2018","unstructured":"Tuomas Haarnoja, Aurick Zhou, Pieter Abbeel, and Sergey Levine. 2018. Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. In International conference on machine learning. PMLR."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467199"},{"key":"e_1_3_2_2_17_1","volume-title":"Yee Whye Teh, and Nicolas Heess","author":"Humplik Jan","year":"2019","unstructured":"Jan Humplik, Alexandre Galashov, Leonard Hasenclever, Pedro A Ortega, Yee Whye Teh, and Nicolas Heess. 2019. Meta reinforcement learning as task inference. arXiv preprint arXiv:1905.06424 (2019)."},{"key":"e_1_3_2_2_18_1","volume-title":"International Conference on Machine Learning. PMLR, 2117--2126","author":"Igl Maximilian","year":"2018","unstructured":"Maximilian Igl, Luisa Zintgraf, Tuan Anh Le, FrankWood, and Shimon Whiteson. 2018. Deep variational reinforcement learning for POMDPs. In International Conference on Machine Learning. PMLR, 2117--2126."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.3390\/e22010045"},{"key":"e_1_3_2_2_20_1","volume-title":"Cassandra","author":"Kaelbling Leslie Pack","year":"1998","unstructured":"Leslie Pack Kaelbling, Michael L. Littman, and Anthony R. Cassandra. 1998. Planning and acting in partially observable stochastic domains. Artificial Intelligence (1998)."},{"key":"e_1_3_2_2_21_1","volume-title":"Canadian Conference on Artificial Intelligence. Springer, 325--330","author":"Karpathy Andrej","unstructured":"Andrej Karpathy and Michiel van de Panne. 2012. Curriculum learning for motor skills. In Canadian Conference on Artificial Intelligence. Springer, 325--330."},{"key":"e_1_3_2_2_22_1","volume-title":"Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114","author":"Kingma Diederik P","year":"2013","unstructured":"Diederik P Kingma and Max Welling. 2013. Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 (2013)."},{"key":"e_1_3_2_2_23_1","volume-title":"Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971","author":"Lillicrap Timothy P","year":"2015","unstructured":"Timothy P Lillicrap, Jonathan J Hunt, Alexander Pritzel, Nicolas Heess, Tom Erez, Yuval Tassa, David Silver, and Daan Wierstra. 2015. Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971 (2015)."},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2983323.2983656"},{"key":"e_1_3_2_2_25_1","volume-title":"Bayesian decision problems and Markov chains","author":"Martin James John","unstructured":"James John Martin. 1967. Bayesian decision problems and Markov chains. Wiley."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-335-6.50030-1"},{"key":"e_1_3_2_2_27_1","volume-title":"Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602","author":"Mnih Volodymyr","year":"2013","unstructured":"Volodymyr Mnih, Koray Kavukcuoglu, David Silver, Alex Graves, Ioannis Antonoglou, Daan Wierstra, and Martin Riedmiller. 2013. Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602 (2013)."},{"key":"e_1_3_2_2_28_1","volume-title":"Advances in Neural Information Processing Systems 26","author":"Osband Ian","year":"2013","unstructured":"Ian Osband, Daniel Russo, and Benjamin Van Roy. 2013. (More) efficient reinforcement learning via posterior sampling. Advances in Neural Information Processing Systems 26 (2013)."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.70"},{"key":"e_1_3_2_2_30_1","volume-title":"ICML","volume":"2000","author":"Strens Malcolm","year":"2000","unstructured":"Malcolm Strens. 2000. A Bayesian framework for reinforcement learning. In ICML, Vol. 2000. 943--950."},{"key":"e_1_3_2_2_31_1","volume-title":"Reinforcement learning: An introduction","author":"Sutton Richard S","unstructured":"Richard S Sutton and Andrew G Barto. 2018. Reinforcement learning: An introduction. MIT press."},{"key":"e_1_3_2_2_32_1","volume-title":"Reward constrained policy optimization. arXiv preprint arXiv:1805.11074","author":"Tessler Chen","year":"2018","unstructured":"Chen Tessler, Daniel J Mankowitz, and Shie Mannor. 2018. Reward constrained policy optimization. arXiv preprint arXiv:1805.11074 (2018)."},{"key":"e_1_3_2_2_33_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.5555\/3398761.3398927"},{"key":"e_1_3_2_2_35_1","first-page":"1","article-title":"A Revenue-Maximizing Bidding Strategy for Demand-Side Platforms","volume":"99","author":"Wang T.","year":"2019","unstructured":"T. Wang, H. Yang, H. Yu, W. Zhou, and H. Song. 2019. A Revenue-Maximizing Bidding Strategy for Demand-Side Platforms. IEEE Access PP, 99 (2019), 1--1.","journal-title":"IEEE Access PP"},{"key":"e_1_3_2_2_36_1","unstructured":"Christopher A Wilkens Ruggiero Cavallo Rad Niazadeh and Samuel Taggart. 2016. Mechanism Design for Value Maximizers."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"crossref","unstructured":"D. Wu X. Chen X. Yang H. Wang Q. Tan X. Zhang J. Xu and K. Gai. 2018. Budget Constrained Bidding by Model-free Reinforcement Learning in Display Advertising. ACM (2018).","DOI":"10.1145\/3269206.3271748"},{"key":"e_1_3_2_2_38_1","unstructured":"Yuxin Wu and Yuandong Tian. 2016. Training agent for first-person shooter game with actor-critic curriculum learning. (2016)."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330681"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/2501040.2501980"},{"key":"e_1_3_2_2_41_1","volume-title":"Optimal Real-Time Bidding for Display Advertising","author":"Zhang W.","unstructured":"W. Zhang. 2016. Optimal Real-Time Bidding for Display Advertising. In UCL (University College London)."},{"key":"e_1_3_2_2_42_1","volume-title":"4th International Workshop, WINE 2008, Shanghai, China, December 17--20, 2008. Proceedings.","author":"Zhou Y.","year":"2008","unstructured":"Y. Zhou, D. Chakrabarty, and Rajan M Lukose. 2008. Budget constrained bidding in keyword auctions and online knapsack problems. In Internet and Network Economics, 4th International Workshop, WINE 2008, Shanghai, China, December 17--20, 2008. Proceedings."},{"key":"e_1_3_2_2_43_1","volume-title":"Varibad: A very good method for bayes-adaptive deep rl via meta-learning. arXiv preprint arXiv:1910.08348","author":"Zintgraf Luisa","year":"2019","unstructured":"Luisa Zintgraf, Kyriacos Shiarlis, Maximilian Igl, Sebastian Schulze, Yarin Gal, Katja Hofmann, and Shimon Whiteson. 2019. Varibad: A very good method for bayes-adaptive deep rl via meta-learning. arXiv preprint arXiv:1910.08348 (2019)."}],"event":{"name":"KDD '22: The 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Washington DC USA","acronym":"KDD '22","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3534678.3539211","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3534678.3539211","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T18:59:58Z","timestamp":1750186798000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3534678.3539211"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,14]]},"references-count":42,"alternative-id":["10.1145\/3534678.3539211","10.1145\/3534678"],"URL":"https:\/\/doi.org\/10.1145\/3534678.3539211","relation":{},"subject":[],"published":{"date-parts":[[2022,8,14]]},"assertion":[{"value":"2022-08-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}