{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T16:02:47Z","timestamp":1780329767658,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":23,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,7]],"date-time":"2026-06-07T00:00:00Z","timestamp":1780790400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,8]]},"DOI":"10.1145\/3774935.3807904","type":"proceedings-article","created":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T14:44:13Z","timestamp":1780325053000},"page":"599-602","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Uncertainty-Aware Reinforcement Learning for Conversion-Optimized Content Gating"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-0499-1245","authenticated-orcid":false,"given":"Han Jun","family":"Yoon","sequence":"first","affiliation":[{"name":"The Washington Post, DC, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-1562-4211","authenticated-orcid":false,"given":"Janith","family":"Weerasinghe","sequence":"additional","affiliation":[{"name":"The Washington Post, DC, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7822-3463","authenticated-orcid":false,"given":"Himanshu","family":"Jahagirdar","sequence":"additional","affiliation":[{"name":"The Washington Post, DC, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6597-5448","authenticated-orcid":false,"given":"Meng","family":"Ling","sequence":"additional","affiliation":[{"name":"The Washington Post, DC, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1881-436X","authenticated-orcid":false,"given":"Suja","family":"Thomas","sequence":"additional","affiliation":[{"name":"The Washington Post, DC, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1308-3154","authenticated-orcid":false,"given":"Anuradha","family":"Uduwage","sequence":"additional","affiliation":[{"name":"The Washington Post, DC, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2988-1284","authenticated-orcid":false,"given":"Sam","family":"Han","sequence":"additional","affiliation":[{"name":"The Washington Post, DC, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,7]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"crossref","unstructured":"Amar Bhide Prakesh\u00a0S Shah and Ganesh Acharya. 2018. A simplified guide to randomized controlled trials. Acta obstetricia et gynecologica Scandinavica 97 4 (2018) 380\u2013387.","DOI":"10.1111\/aogs.13309"},{"key":"e_1_3_3_2_3_2","first-page":"208","volume-title":"Proceedings of the fourteenth international conference on artificial intelligence and statistics","author":"Chu Wei","year":"2011","unstructured":"Wei Chu, Lihong Li, Lev Reyzin, and Robert Schapire. 2011. Contextual bandits with linear payoff functions. In Proceedings of the fourteenth international conference on artificial intelligence and statistics. JMLR Workshop and Conference Proceedings, 208\u2013214."},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219892"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"crossref","unstructured":"Heidar Davoudi Zana Rashidi Aijun An Morteza Zihayat and Gordon Edall. 2020. Paywall policy learning in digital news media. IEEE Transactions on Knowledge and Data Engineering 33 10 (2020) 3394\u20133409.","DOI":"10.1109\/TKDE.2020.2969419"},{"key":"e_1_3_3_2_6_2","first-page":"2052","volume-title":"International Conference on Machine Learning (ICML)","author":"Fujimoto Scott","year":"2019","unstructured":"Scott Fujimoto, David Meger, and Doina Precup. 2019. Off-Policy Deep Reinforcement Learning without Exploration. In International Conference on Machine Learning (ICML). 2052\u20132062."},{"key":"e_1_3_3_2_7_2","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Fujimoto Scott","year":"2021","unstructured":"Scott Fujimoto, David Meger, and Doina Precup. 2021. A Minimalist Approach to Offline Reinforcement Learning. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_2_8_2","first-page":"1861","volume-title":"International Conference on Machine Learning (ICML)","author":"Haarnoja Tuomas","year":"2018","unstructured":"Tuomas Haarnoja, Aurick Zhou, Pieter Abbeel, and Sergey Levine. 2018. Soft Actor-Critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. In International Conference on Machine Learning (ICML). 1861\u20131870."},{"key":"e_1_3_3_2_9_2","volume-title":"International Conference on Learning Representations (ICLR)","author":"Kostrikov Ilya","year":"2022","unstructured":"Ilya Kostrikov, Ashvin Nair, and Sergey Levine. 2022. Offline Reinforcement Learning with Implicit Q-Learning. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_2_10_2","first-page":"11784","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Kumar Aviral","year":"2019","unstructured":"Aviral Kumar, Justin Fu, George Tucker, and Sergey Levine. 2019. Stabilizing Off-Policy Q-Learning via Bootstrapping Error Reduction. In Advances in Neural Information Processing Systems (NeurIPS). 11784\u201311794."},{"key":"e_1_3_3_2_11_2","first-page":"1179","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Kumar Aviral","year":"2020","unstructured":"Aviral Kumar, Aurick Zhou, George Tucker, and Sergey Levine. 2020. Conservative Q-Learning for Offline Reinforcement Learning. In Advances in Neural Information Processing Systems (NeurIPS). 1179\u20131191."},{"key":"e_1_3_3_2_12_2","unstructured":"Sergey Levine Aviral Kumar George Tucker and Justin Fu. 2020. Offline reinforcement learning: Tutorial review and perspectives on open problems. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2005.01643 (2020)."},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/1772690.1772758"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","unstructured":"Volodymyr Mnih Koray Kavukcuoglu David Silver et\u00a0al. 2015. Human-level control through deep reinforcement learning. Nature 518 7540 (2015) 529\u2013533.","DOI":"10.1038\/nature14236"},{"key":"e_1_3_3_2_15_2","first-page":"753","volume-title":"Conference on Robot Learning (CoRL)","author":"Nair Ashvin","year":"2020","unstructured":"Ashvin Nair, Abhishek Gupta, Murtaza Dalal, and Sergey Levine. 2020. Accelerating Online Reinforcement Learning with Offline Datasets. In Conference on Robot Learning (CoRL). 753\u2013767."},{"key":"e_1_3_3_2_16_2","unstructured":"Gergely Neu and Ciara Pike-Burke. 2020. A unifying view of optimism in episodic reinforcement learning. Advances in Neural Information Processing Systems 33 (2020) 1392\u20131403."},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"crossref","unstructured":"Rafael\u00a0Figueiredo Prudencio Marcos\u00a0ROA Maximo and Esther\u00a0Luna Colombini. 2023. A survey on offline reinforcement learning: Taxonomy review and open problems. IEEE Transactions on Neural Networks and Learning Systems 35 8 (2023) 10237\u201310257.","DOI":"10.1109\/TNNLS.2023.3250269"},{"key":"e_1_3_3_2_18_2","volume-title":"Markov decision processes: discrete stochastic dynamic programming","author":"Puterman Martin\u00a0L","year":"2014","unstructured":"Martin\u00a0L Puterman. 2014. Markov decision processes: discrete stochastic dynamic programming. John Wiley & Sons."},{"key":"e_1_3_3_2_19_2","first-page":"1889","volume-title":"International Conference on Machine Learning (ICML)","author":"Schulman John","year":"2015","unstructured":"John Schulman, Sergey Levine, Philipp Moritz, Michael Jordan, and Pieter Abbeel. 2015. Trust Region Policy Optimization. In International Conference on Machine Learning (ICML). 1889\u20131897."},{"key":"e_1_3_3_2_20_2","volume-title":"arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1707.06347","author":"Schulman John","year":"2017","unstructured":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov. 2017. Proximal Policy Optimization Algorithms. In arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1707.06347."},{"key":"e_1_3_3_2_21_2","first-page":"2413","volume-title":"International Conference on Artificial Intelligence and Statistics","author":"Sondhi Arjun","year":"2020","unstructured":"Arjun Sondhi, David Arbour, and Drew Dimmery. 2020. Balanced off-policy evaluation in general action spaces. In International Conference on Artificial Intelligence and Statistics. PMLR, 2413\u20132423."},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","DOI":"10.5555\/551283"},{"key":"e_1_3_3_2_23_2","unstructured":"Nino Vieillard Tadashi Kozuno Bruno Scherrer Olivier Pietquin R\u00e9mi Munos and Matthieu Geist. 2020. Leverage the average: an analysis of kl regularization in reinforcement learning. Advances in Neural Information Processing Systems 33 (2020) 12163\u201312174."},{"key":"e_1_3_3_2_24_2","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Wang Ziyu","year":"2020","unstructured":"Ziyu Wang, Alexander Novikov, Konrad Zolna, et\u00a0al. 2020. Critic Regularized Regression. In Advances in Neural Information Processing Systems (NeurIPS)."}],"event":{"name":"UMAP '26: 34th ACM Conference on User Modeling, Adaptation and Personalization","location":"Gothenburg , Sweden","acronym":"UMAP '26","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 34th ACM Conference on User Modeling, Adaptation and Personalization"],"original-title":[],"deposited":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T15:14:02Z","timestamp":1780326842000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774935.3807904"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,7]]},"references-count":23,"alternative-id":["10.1145\/3774935.3807904","10.1145\/3774935"],"URL":"https:\/\/doi.org\/10.1145\/3774935.3807904","relation":{},"subject":[],"published":{"date-parts":[[2026,6,7]]},"assertion":[{"value":"2026-06-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}