{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T16:02:16Z","timestamp":1780329736983,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":7,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,7]],"date-time":"2026-06-07T00:00:00Z","timestamp":1780790400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,8]]},"DOI":"10.1145\/3774935.3807911","type":"proceedings-article","created":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T14:44:13Z","timestamp":1780325053000},"page":"571-574","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["A Case Study of Offline Reinforcement Learning for Paywall Decisioning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-1562-4211","authenticated-orcid":false,"given":"Janith","family":"Weerasinghe","sequence":"first","affiliation":[{"name":"The Washington Post, Washington, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-0499-1245","authenticated-orcid":false,"given":"Han Jun","family":"Yoon","sequence":"additional","affiliation":[{"name":"The Washington Post, Washington, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6597-5448","authenticated-orcid":false,"given":"Meng","family":"Ling","sequence":"additional","affiliation":[{"name":"The Washington Post, Washington, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7822-3463","authenticated-orcid":false,"given":"Himanshu","family":"Jahagirdar","sequence":"additional","affiliation":[{"name":"The Washington Post, Washington, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1881-436X","authenticated-orcid":false,"given":"Suja","family":"Thomas","sequence":"additional","affiliation":[{"name":"The Washington Post, Washington, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1308-3154","authenticated-orcid":false,"given":"Anuradha","family":"Uduwage","sequence":"additional","affiliation":[{"name":"The Washington Post, Washington, DC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2988-1284","authenticated-orcid":false,"given":"Sam","family":"Han","sequence":"additional","affiliation":[{"name":"The Washington Post, Washington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,7]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219892"},{"key":"e_1_3_3_1_3_2","series-title":"(NIPS \u201920)","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","author":"Kumar Aviral","year":"2020","unstructured":"Aviral Kumar, Aurick Zhou, George Tucker, and Sergey Levine. 2020. Conservative Q-learning for offline reinforcement learning. In Proceedings of the 34th International Conference on Neural Information Processing Systems (Vancouver, BC, Canada) (NIPS \u201920). Curran Associates Inc., Red Hook, NY, USA, Article 100, 13\u00a0pages."},{"key":"e_1_3_3_1_4_2","first-page":"3703","volume-title":"International Conference on Machine Learning (ICML)","author":"Le Hoang","year":"2019","unstructured":"Hoang Le, Cameron Voloshin, and Yisong Yue. 2019. Batch Policy Learning under Constraints. In International Conference on Machine Learning (ICML). PMLR, 3703\u20133712."},{"key":"e_1_3_3_1_5_2","unstructured":"Takuma Seno and Michita Imai. 2022. d3rlpy: An Offline Deep Reinforcement Learning Library. Journal of Machine Learning Research 23 315 (2022) 1\u201320. http:\/\/jmlr.org\/papers\/v23\/22-0017.html"},{"key":"e_1_3_3_1_6_2","unstructured":"Rohit Supekar. 2022. How The New York Times Uses Machine Learning To Make Its Paywall Smarter. https:\/\/open.nytimes.com\/how-the-new-york-times-uses-machine-learning-to-make-its-paywall-smarter-e5771d5f46f8 Accessed: 2026-03-06."},{"key":"e_1_3_3_1_7_2","unstructured":"Rohit Supekar. 2025. Scaling Subscriptions at The New York Times with Real-Time Causal Machine Learning. https:\/\/open.nytimes.com\/scaling-subscriptions-at-the-new-york-times-with-real-time-causal-machine-learning-5f23a7b24ff4 Accessed: 2026-03-06."},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","unstructured":"Philip Thomas Georgios Theocharous and Mohammad Ghavamzadeh. 2015. High-Confidence Off-Policy Evaluation. Proceedings of the AAAI Conference on Artificial Intelligence 29 1 (Feb. 2015). 10.1609\/aaai.v29i1.9541","DOI":"10.1609\/aaai.v29i1.9541"}],"event":{"name":"UMAP '26: 34th ACM Conference on User Modeling, Adaptation and Personalization","location":"Gothenburg , Sweden","acronym":"UMAP '26","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 34th ACM Conference on User Modeling, Adaptation and Personalization"],"original-title":[],"deposited":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T15:12:27Z","timestamp":1780326747000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774935.3807911"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,7]]},"references-count":7,"alternative-id":["10.1145\/3774935.3807911","10.1145\/3774935"],"URL":"https:\/\/doi.org\/10.1145\/3774935.3807911","relation":{},"subject":[],"published":{"date-parts":[[2026,6,7]]},"assertion":[{"value":"2026-06-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}