{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T13:06:02Z","timestamp":1785503162509,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","funder":[{"name":"National Natural Science Foundation of China","award":["62272349"],"award-info":[{"award-number":["62272349"]}]},{"name":"National Natural Science Foundation of China","award":["62302345"],"award-info":[{"award-number":["62302345"]}]},{"name":"National Natural Science Foundation of China","award":["U23A20305"],"award-info":[{"award-number":["U23A20305"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,8,9]]},"DOI":"10.1145\/3770854.3783950","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T12:07:40Z","timestamp":1785499660000},"page":"2573-2583","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Q-Regularized Generative Auto-Bidding: From Suboptimal Trajectories to Optimal Policies"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-1730-4584","authenticated-orcid":false,"given":"Mingming","family":"Zhang","sequence":"first","affiliation":[{"name":"Key Laboratory of Aerospace Information Security and Trusted Computing, Ministry of Education, School of Cyber Science and Engineering, Wuhan University Taobao &amp;#38; Tmall Group of Alibaba, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-3394-0056","authenticated-orcid":false,"given":"Na","family":"Li","sequence":"additional","affiliation":[{"name":"Key Laboratory of Aerospace Information Security and Trusted Computing, Ministry of Education, School of Cyber Science and Engineering, Wuhan University Taobao &amp;#38; Tmall Group of Alibaba, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-3256-497X","authenticated-orcid":false,"given":"Feiqing","family":"Zhuang","sequence":"additional","affiliation":[{"name":"Taobao &amp;#38; Tmall Group of Alibaba, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7376-1991","authenticated-orcid":false,"given":"Hongyang","family":"Zheng","sequence":"additional","affiliation":[{"name":"Taobao &amp;#38; Tmall Group of Alibaba, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-7743-6644","authenticated-orcid":false,"given":"Jiangbing","family":"Zhou","sequence":"additional","affiliation":[{"name":"Taobao &amp;#38; Tmall Group of Alibaba, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-4944-4817","authenticated-orcid":false,"given":"Wuyin","family":"Wang","sequence":"additional","affiliation":[{"name":"Taobao &amp;#38; Tmall Group of Alibaba, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1519-6682","authenticated-orcid":false,"given":"Shengjie","family":"Sun","sequence":"additional","affiliation":[{"name":"Taobao &amp;#38; Tmall Group of Alibaba, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2175-1240","authenticated-orcid":false,"given":"Xiaowei","family":"Chen","sequence":"additional","affiliation":[{"name":"Taobao &amp;#38; Tmall Group of Alibaba, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-4350-5324","authenticated-orcid":false,"given":"Junxiong","family":"Zhu","sequence":"additional","affiliation":[{"name":"Taobao &amp;#38; Tmall Group of Alibaba, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6755-871X","authenticated-orcid":false,"given":"Lixin","family":"Zou","sequence":"additional","affiliation":[{"name":"Key Laboratory of Aerospace Information Security and Trusted Computing, Ministry of Education, School of Cyber Science and Engineering, Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3144-6374","authenticated-orcid":false,"given":"Chenliang","family":"Li","sequence":"additional","affiliation":[{"name":"Key Laboratory of Aerospace Information Security and Trusted Computing, Ministry of Education, School of Cyber Science and Engineering, Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,20]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3699824.3699838"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"crossref","unstructured":"Michael Bain and Claude Sammut. 1995. A Framework for Behavioural Cloning.. In Machine intelligence 15. 103-129.","DOI":"10.1093\/oso\/9780198538677.003.0006"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1090\/S0002-9904-1954-09848-8"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-009-9112-y"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2012.2186810"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3018661.3018702"},{"key":"e_1_3_2_2_7_1","first-page":"104","volume-title":"RTBAgent: A LLM-based Agent System for Real-Time Bidding. In Companion Proceedings of the ACM on Web Conference","author":"Cai Leng","year":"2025","unstructured":"Leng Cai, Junxuan He, Yikai Li, Junjie Liang, Yuanping Lin, Ziming Quan, Yawen Zeng, and Jin Xu. 2025. RTBAgent: A LLM-based Agent System for Real-Time Bidding. In Companion Proceedings of the ACM on Web Conference 2025. 104-113."},{"key":"e_1_3_2_2_8_1","volume-title":"Decision transformer: Reinforcement learning via sequence modeling. Advances in neural information processing systems","author":"Chen Lili","year":"2021","unstructured":"Lili Chen, Kevin Lu, Aravind Rajeswaran, Kimin Lee, Aditya Grover, Misha Laskin, Pieter Abbeel, Aravind Srinivas, and Igor Mordatch. 2021. Decision transformer: Reinforcement learning via sequence modeling. Advances in neural information processing systems, Vol. 34 (2021), 15084-15097."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/2020408.2020604"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2015.12.114"},{"key":"e_1_3_2_2_11_1","volume-title":"Internet advertising and the generalized second-price auction: Selling billions of dollars worth of keywords. American economic review","author":"Edelman Benjamin","year":"2007","unstructured":"Benjamin Edelman, Michael Ostrovsky, and Michael Schwarz. 2007. Internet advertising and the generalized second-price auction: Selling billions of dollars worth of keywords. American economic review, Vol. 97, 1 (2007), 242-259."},{"key":"e_1_3_2_2_12_1","volume-title":"A minimalist approach to offline reinforcement learning. Advances in neural information processing systems","author":"Fujimoto Scott","year":"2021","unstructured":"Scott Fujimoto and Shixiang Shane Gu. 2021. A minimalist approach to offline reinforcement learning. Advances in neural information processing systems, Vol. 34 (2021), 20132-20145."},{"key":"e_1_3_2_2_13_1","volume-title":"International conference on machine learning. PMLR, 1587-1596","author":"Fujimoto Scott","year":"2018","unstructured":"Scott Fujimoto, Herke Hoof, and David Meger. 2018. Addressing function approximation error in actor-critic methods. In International conference on machine learning. PMLR, 1587-1596."},{"key":"e_1_3_2_2_14_1","volume-title":"International conference on machine learning. PMLR","author":"Fujimoto Scott","year":"2019","unstructured":"Scott Fujimoto, David Meger, and Doina Precup. 2019. Off-policy deep reinforcement learning without exploration. In International conference on machine learning. PMLR, 2052-2062."},{"key":"e_1_3_2_2_15_1","unstructured":"Jingtong Gao Yewen Li Shuai Mao Peng Jiang Nan Jiang Yejing Wang Qingpeng Cai Fei Pan Kun Gai Bo An et al. 2025. Generative Auto-Bidding with Value-Guided Explorations. arXiv preprint arXiv:2504.14587 (2025)."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671526"},{"key":"e_1_3_2_2_17_1","volume-title":"Advances in neural information processing systems","author":"Hasselt Hado","year":"2010","unstructured":"Hado Hasselt. 2010. Double Q-learning. Advances in neural information processing systems, Vol. 23 (2010)."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467199"},{"key":"e_1_3_2_2_19_1","volume-title":"Model-based trajectory stitching for improved offline reinforcement learning. arXiv preprint arXiv:2211.11603","author":"Hepburn Charles A","year":"2022","unstructured":"Charles A Hepburn and Giovanni Montana. 2022. Model-based trajectory stitching for improved offline reinforcement learning. arXiv preprint arXiv:2211.11603 (2022)."},{"key":"e_1_3_2_2_20_1","volume-title":"Q-value regularized transformer for offline reinforcement learning. arXiv preprint arXiv:2405.17098","author":"Hu Shengchao","year":"2024","unstructured":"Shengchao Hu, Ziqing Fan, Chaoqin Huang, Li Shen, Ya Zhang, Yanfeng Wang, and Dacheng Tao. 2024. Q-value regularized transformer for offline reinforcement learning. arXiv preprint arXiv:2405.17098 (2024)."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2024.04.191"},{"key":"e_1_3_2_2_22_1","first-page":"1033","volume-title":"Optimal Return-to-Go Guided Decision Transformer for Auto-Bidding in Advertisement. In Companion Proceedings of the ACM on Web Conference","author":"Jiang Hao","year":"2025","unstructured":"Hao Jiang, Yongxiang Tang, Yanxiang Zeng, Pengjia Yuan, Yanhua Cheng, Teng Sha, Xialong Liu, and Peng Jiang. 2025. Optimal Return-to-Go Guided Decision Transformer for Auto-Bidding in Advertisement. In Companion Proceedings of the ACM on Web Conference 2025. 1033-1037."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671555"},{"key":"e_1_3_2_2_24_1","volume-title":"Offline reinforcement learning with implicit q-learning. arXiv preprint arXiv:2110.06169","author":"Kostrikov Ilya","year":"2021","unstructured":"Ilya Kostrikov, Ashvin Nair, and Sergey Levine. 2021. Offline reinforcement learning with implicit q-learning. arXiv preprint arXiv:2110.06169 (2021)."},{"key":"e_1_3_2_2_25_1","volume-title":"Conservative q-learning for offline reinforcement learning. Advances in neural information processing systems","author":"Kumar Aviral","year":"2020","unstructured":"Aviral Kumar, Aurick Zhou, George Tucker, and Sergey Levine. 2020. Conservative q-learning for offline reinforcement learning. Advances in neural information processing systems, Vol. 33 (2020), 1179-1191."},{"key":"e_1_3_2_2_26_1","volume-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems. arXiv preprint arXiv:2005.01643","author":"Levine Sergey","year":"2020","unstructured":"Sergey Levine, Aviral Kumar, George Tucker, and Justin Fu. 2020. Offline reinforcement learning: Tutorial, review, and perspectives on open problems. arXiv preprint arXiv:2005.01643 (2020)."},{"key":"e_1_3_2_2_27_1","volume-title":"Diffstitch: Boosting offline reinforcement learning with diffusion-based trajectory stitching. arXiv preprint arXiv:2402.02439","author":"Li Guanghe","year":"2024","unstructured":"Guanghe Li, Yixiang Shan, Zhengbang Zhu, Ting Long, and Weinan Zhang. 2024b. Diffstitch: Boosting offline reinforcement learning with diffusion-based trajectory stitching. arXiv preprint arXiv:2402.02439 (2024)."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645534"},{"key":"e_1_3_2_2_29_1","volume-title":"Auto-bidding equilibrium in ROI-constrained online advertising markets. arXiv preprint arXiv:2210.06107","author":"Li Juncheng","year":"2022","unstructured":"Juncheng Li and Pingzhong Tang. 2022. Auto-bidding equilibrium in ROI-constrained online advertising markets. arXiv preprint arXiv:2210.06107 (2022)."},{"key":"e_1_3_2_2_30_1","first-page":"1104","volume-title":"EBaReT: Expert-guided Bag Reward Transformer for Auto Bidding. In Companion Proceedings of the ACM on Web Conference","author":"Li Kaiyuan","year":"2025","unstructured":"Kaiyuan Li, Pengyu Wang, Yunshan Peng, Pengjia Yuan, Yanxiang Zeng, Rui Xiang, Yanhua Cheng, Xialong Liu, and Peng Jiang. 2025b. EBaReT: Expert-guided Bag Reward Transformer for Auto Bidding. In Companion Proceedings of the ACM on Web Conference 2025. 1104-1108."},{"key":"e_1_3_2_2_31_1","first-page":"315","volume-title":"GAS: Generative Auto-bidding with Post-training Search. In Companion Proceedings of the ACM on Web Conference","author":"Li Yewen","year":"2025","unstructured":"Yewen Li, Shuai Mao, Jingtong Gao, Nan Jiang, Yunjian Xu, Qingpeng Cai, Fei Pan, Peng Jiang, and Bo An. 2025a. GAS: Generative Auto-bidding with Post-training Search. In Companion Proceedings of the ACM on Web Conference 2025. 315-324."},{"key":"e_1_3_2_2_32_1","volume-title":"Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971","author":"Lillicrap Timothy P","year":"2015","unstructured":"Timothy P Lillicrap, Jonathan J Hunt, Alexander Pritzel, Nicolas Heess, Tom Erez, Yuval Tassa, David Silver, and Daan Wierstra. 2015. Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971 (2015)."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.3037940"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.2970463"},{"key":"e_1_3_2_2_35_1","volume-title":"International conference on machine learning. PMLR, 21611-21630","author":"Liu Zuxin","year":"2023","unstructured":"Zuxin Liu, Zijian Guo, Yihang Yao, Zhepeng Cen, Wenhao Yu, Tingnan Zhang, and Ding Zhao. 2023. Constrained decision transformer for offline safe reinforcement learning. In International conference on machine learning. PMLR, 21611-21630."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"crossref","unstructured":"Volodymyr Mnih Koray Kavukcuoglu David Silver Andrei A Rusu Joel Veness Marc G Bellemare Alex Graves Martin Riedmiller Andreas K Fidjeland Georg Ostrovski et al. 2015. Human-level control through deep reinforcement learning. nature Vol. 518 7540 (2015) 529-533.","DOI":"10.1038\/nature14236"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0192"},{"key":"e_1_3_2_2_38_1","volume-title":"International conference on machine learning. PMLR","author":"Schulman John","year":"2015","unstructured":"John Schulman, Sergey Levine, Pieter Abbeel, Michael Jordan, and Philipp Moritz. 2015. Trust region policy optimization. In International conference on machine learning. PMLR, 1889-1897."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2993"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-022-10228-y"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"e_1_3_2_2_42_1","volume-title":"Yi Wong, Ziru Liu, Xiangyu Zhao, Yichao Wang, Bo Chen, Huifeng Guo, and Ruiming Tang.","author":"Wang Yuhao","year":"2023","unstructured":"Yuhao Wang, Ha Tsz Lam, Yi Wong, Ziru Liu, Xiangyu Zhao, Yichao Wang, Bo Chen, Huifeng Guo, and Ruiming Tang. 2023. Multi-task deep recommender systems: A survey. arXiv preprint arXiv:2302.03525 (2023)."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3488560.3498373"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3269206.3271748"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330681"},{"key":"e_1_3_2_2_46_1","volume-title":"Advances in Neural Information Processing Systems","volume":"30","author":"Yu Hao","year":"2017","unstructured":"Hao Yu, Michael Neely, and Xiaohan Wei. 2017. Online convex optimization with stochastic constraints. Advances in Neural Information Processing Systems, Vol. 30 (2017)."},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.5555\/3455716.3455717"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3511808.3557064"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599765"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/2835776.2835843"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219918"}],"event":{"name":"KDD '26: The 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Jeju Island Republic of Korea","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.1"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3770854.3783950","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T12:26:49Z","timestamp":1785500809000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3770854.3783950"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,20]]},"references-count":51,"alternative-id":["10.1145\/3770854.3783950","10.1145\/3770854"],"URL":"https:\/\/doi.org\/10.1145\/3770854.3783950","relation":{},"subject":[],"published":{"date-parts":[[2026,4,20]]},"assertion":[{"value":"2026-04-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}