{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T00:27:34Z","timestamp":1787012854325,"version":"build-2736575974"},"reference-count":44,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2024,3,1]],"date-time":"2024-03-01T00:00:00Z","timestamp":1709251200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,3,1]],"date-time":"2024-03-01T00:00:00Z","timestamp":1709251200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,3,1]],"date-time":"2024-03-01T00:00:00Z","timestamp":1709251200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U23A20310"],"award-info":[{"award-number":["U23A20310"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U21A20519"],"award-info":[{"award-number":["U21A20519"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Comput."],"published-print":{"date-parts":[[2024,3]]},"DOI":"10.1109\/tc.2023.3343111","type":"journal-article","created":{"date-parts":[[2023,12,14]],"date-time":"2023-12-14T14:25:57Z","timestamp":1702563957000},"page":"815-828","source":"Crossref","is-referenced-by-count":7,"title":["HiBid: A Cross-Channel Constrained Bidding System With Budget Allocation by Hierarchical Offline Deep Reinforcement Learning"],"prefix":"10.1109","volume":"73","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-0199-0488","authenticated-orcid":false,"given":"Hao","family":"Wang","sequence":"first","affiliation":[{"name":"School of Computer Science and Technology, Beijing Institute of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7129-0250","authenticated-orcid":false,"given":"Bo","family":"Tang","sequence":"additional","affiliation":[{"name":"Meituan, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0252-329X","authenticated-orcid":false,"given":"Chi Harold","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Beijing Institute of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3247-0483","authenticated-orcid":false,"given":"Shangqin","family":"Mao","sequence":"additional","affiliation":[{"name":"Meituan, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1319-2369","authenticated-orcid":false,"given":"Jiahong","family":"Zhou","sequence":"additional","affiliation":[{"name":"Meituan, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2479-9801","authenticated-orcid":false,"given":"Zipeng","family":"Dai","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Beijing Institute of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3961-7652","authenticated-orcid":false,"given":"Yaqi","family":"Sun","sequence":"additional","affiliation":[{"name":"Meituan, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5400-7924","authenticated-orcid":false,"given":"Qianlong","family":"Xie","sequence":"additional","affiliation":[{"name":"Meituan, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5495-0827","authenticated-orcid":false,"given":"Xingxing","family":"Wang","sequence":"additional","affiliation":[{"name":"Meituan, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1964-3984","authenticated-orcid":false,"given":"Dong","family":"Wang","sequence":"additional","affiliation":[{"name":"Meituan, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2023.3251850"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2015.2435784"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2014.2346204"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/2020408.2020604"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1257\/aer.99.2.430"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3097983.3098134"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3018661.3018702"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539211"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3357384.3358031"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/2187836.2187888"},{"key":"ref11","volume-title":"Operations Research: An Introduction","volume":"7","author":"Taha","year":"2003"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1201\/9781315140223"},{"key":"ref13","article-title":"Opal: Offline primitive discovery for accelerating offline reinforcement learning","volume-title":"Proc. ICLR","author":"Ajay","year":"2021"},{"key":"ref14","first-page":"11553","article-title":"Hierarchical skills for efficient exploration","volume-title":"Proc. NeurIPS","volume":"34","author":"Gehring","year":"2021"},{"key":"ref15","first-page":"1711","article-title":"Mildly conservative Q-learning for offline reinforcement learning","volume-title":"Proc. NeurIPS","author":"Lyu","year":"2022"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3512109"},{"key":"ref17","first-page":"20410","article-title":"Bcorle($\\lambda$\u03bb): An offline reinforcement learning and evaluation framework for coupons allocation in e-commerce market","volume-title":"Proc. NeurIPS","volume":"34","author":"Zhang","year":"2021"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/2623330.2623633"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467167"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467113"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2017.2775228"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/3269206.3271748"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330681"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICDM.2019.00122"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i5.16580"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/2983323.2983656"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467199"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219918"},{"key":"ref29","article-title":"Stabilizing off-policy Q-learning via bootstrapping error reduction","volume-title":"Proc. NeurIPS","volume":"32","author":"Kumar","year":"2019"},{"key":"ref30","article-title":"Way off-policy batch deep reinforcement learning of implicit human preferences in dialog","author":"Jaques","year":"2019"},{"key":"ref31","first-page":"2052","article-title":"Off-policy deep reinforcement learning without exploration","volume-title":"Proc. ICML","author":"Fujimoto","year":"2019"},{"issue":"42","key":"ref32","first-page":"1437","article-title":"A comprehensive survey on safe reinforcement learning","volume":"16","author":"Garc\u00eda","year":"2015","journal-title":"J. Mach. Learn. Res."},{"key":"ref33","first-page":"8378","article-title":"Natural policy gradient primal-dual method for constrained Markov decision processes","volume-title":"Proc. NeurIPS","volume":"33","author":"Ding","year":"2020"},{"key":"ref34","first-page":"3703","article-title":"Batch policy learning under constraints","volume-title":"Proc. ICML","author":"Le","year":"2019"},{"key":"ref35","article-title":"Reward constrained policy optimization","volume-title":"Proc. ICLR","author":"Tessler","year":"2019"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref37","first-page":"3483","article-title":"Learning structured output representation using deep conditional generative models","volume-title":"Proc. NeurIPS","volume":"28","author":"Sohn","year":"2015"},{"key":"ref38","first-page":"7555","article-title":"Constrained reinforcement learning has zero duality gap","volume-title":"Proc. NeurIPS","author":"Paternain","year":"2019"},{"key":"ref39","first-page":"28701","article-title":"Policy regularization with dataset constraint for offline reinforcement learning","volume-title":"Proc. of the 40th Internl Conf on Machine Learning","volume":"202","author":"Ran","year":"2023"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1007\/s10479-005-5724-z"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/TCST.2005.847331"},{"key":"ref42","article-title":"Mindopt studio"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2015.2444843"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2015.2409864"}],"container-title":["IEEE Transactions on Computers"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/12\/10431415\/10360353.pdf?arnumber=10360353","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,14]],"date-time":"2024-03-14T04:26:36Z","timestamp":1710390396000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10360353\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,3]]},"references-count":44,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tc.2023.3343111","relation":{},"ISSN":["0018-9340","1557-9956","2326-3814"],"issn-type":[{"value":"0018-9340","type":"print"},{"value":"1557-9956","type":"electronic"},{"value":"2326-3814","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,3]]}}}