{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T15:11:49Z","timestamp":1774969909854,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,4,25]],"date-time":"2022-04-25T00:00:00Z","timestamp":1650844800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,4,25]]},"DOI":"10.1145\/3485447.3512109","type":"proceedings-article","created":{"date-parts":[[2022,4,25]],"date-time":"2022-04-25T05:11:23Z","timestamp":1650863483000},"page":"401-409","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":30,"title":["Cross DQN: Cross Deep Q Network for Ads Allocation in Feed"],"prefix":"10.1145","author":[{"given":"Guogang","family":"Liao","sequence":"first","affiliation":[{"name":"Meituan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ze","family":"Wang","sequence":"additional","affiliation":[{"name":"Meituan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoxu","family":"Wu","sequence":"additional","affiliation":[{"name":"Meituan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaowen","family":"Shi","sequence":"additional","affiliation":[{"name":"Meituan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chuheng","family":"Zhang","sequence":"additional","affiliation":[{"name":"IIIS, Tsinghua University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongkang","family":"Wang","sequence":"additional","affiliation":[{"name":"Meituan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xingxing","family":"Wang","sequence":"additional","affiliation":[{"name":"Meituan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dong","family":"Wang","sequence":"additional","affiliation":[{"name":"Meituan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,4,25]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Constrained Markov decision processes. Vol.\u00a07","author":"Altman Eitan","unstructured":"Eitan Altman. 1999. Constrained Markov decision processes. Vol.\u00a07. CRC Press."},{"key":"e_1_3_2_1_2_1","unstructured":"Carlos Carrion Zenan Wang Harikesh Nair Xianghong Luo Yulin Lei Xiliang Lin Wenlong Chen Qiyu Hu Changping Peng Yongjun Bao and Weipeng\u00a0P. Yan. 2021. Blending Advertising with Organic Content in E-Commerce: A Virtual Bids Optimization Approach. ArXiv abs\/2105.13556(2021)."},{"key":"e_1_3_2_1_3_1","volume-title":"Internet advertising and the generalized second-price auction: Selling billions of dollars worth of keywords. American economic review 97, 1","author":"Edelman Benjamin","year":"2007","unstructured":"Benjamin Edelman, Michael Ostrovsky, and Michael Schwarz. 2007. Internet advertising and the generalized second-price auction: Selling billions of dollars worth of keywords. American economic review 97, 1 (2007), 242\u2013259."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3178876.3186165"},{"key":"e_1_3_2_1_5_1","unstructured":"Yufei Feng Yu Gong Fei Sun Qingwen Liu and Wenwu Ou. 2021. Revisit Recommender System in the Permutation Prospective. ArXiv abs\/2102.12057(2021)."},{"key":"e_1_3_2_1_6_1","volume-title":"GRN: Generative Rerank Network for Context-wise Recommendation. ArXiv abs\/2104.00860(2021).","author":"Feng Yufei","year":"2021","unstructured":"Yufei Feng, Binbin Hu, Yu Gong, Fei Sun, Qingwen Liu, and Wenwu Ou. 2021. GRN: Generative Rerank Network for Context-wise Recommendation. ArXiv abs\/2104.00860(2021)."},{"key":"e_1_3_2_1_7_1","first-page":"1605","article-title":"An Empirical Analysis of Search Engine Advertising","volume":"55","author":"Ghose A.","year":"2009","unstructured":"A. Ghose and Sha Yang. 2009. An Empirical Analysis of Search Engine Advertising: Sponsored Search in Electronic Markets. Manag. Sci. 55(2009), 1605\u20131622.","journal-title":"Sponsored Search in Electronic Markets. Manag. Sci."},{"key":"e_1_3_2_1_8_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980(2014).","author":"Kingma P","year":"2014","unstructured":"Diederik\u00a0P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980(2014)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/2944789.2944791"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3340531.3411952"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1561\/0400000057"},{"key":"e_1_3_2_1_12_1","volume-title":"Human-level control through deep reinforcement learning. nature 518, 7540","author":"Mnih Volodymyr","year":"2015","unstructured":"Volodymyr Mnih, Koray Kavukcuoglu, David Silver, Andrei\u00a0A Rusu, Joel Veness, Marc\u00a0G Bellemare, Alex Graves, Martin Riedmiller, Andreas\u00a0K Fidjeland, Georg Ostrovski, 2015. Human-level control through deep reinforcement learning. nature 518, 7540 (2015), 529\u2013533."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3340531.3412728"},{"key":"e_1_3_2_1_14_1","volume-title":"Introduction to reinforcement learning. Vol.\u00a0135","author":"Sutton S","unstructured":"Richard\u00a0S Sutton, Andrew\u00a0G Barto, 1998. Introduction to reinforcement learning. Vol.\u00a0135. MIT press Cambridge."},{"key":"e_1_3_2_1_15_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. arXiv preprint arXiv:1706.03762(2017)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"crossref","unstructured":"B. Wang Zhaonan Li Jie Tang Kuo Zhang Songcan Chen and Liyun Ru. 2011. Learning to Advertise: How Many Ads Are Enough?. In PAKDD.","DOI":"10.1007\/978-3-642-20847-8_42"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3357384.3357806"},{"key":"e_1_3_2_1_18_1","volume-title":"International conference on machine learning. PMLR","author":"Wang Ziyu","year":"2016","unstructured":"Ziyu Wang, Tom Schaul, Matteo Hessel, Hado Hasselt, Marc Lanctot, and Nando Freitas. 2016. Dueling network architectures for deep reinforcement learning. In International conference on machine learning. PMLR, 1995\u20132003."},{"key":"e_1_3_2_1_19_1","unstructured":"Jianxiong Wei Anxiang Zeng Yueqiu Wu Pengxin Guo Q. Hua and Qingpeng Cai. 2020. Generator and Critic: A Deep Reinforcement Learning Approach for Slate Re-ranking in E-commerce. ArXiv abs\/2005.12206(2020)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i5.16580"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403391"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3184558.3191584"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Mengchen Zhao Z. Li Bo An Haifeng Lu Yifan Yang and Chen Chu. 2018. Impression Allocation for Combating Fraud in E-commerce Via Deep Reinforcement Learning with Action Norm Penalty. In IJCAI.","DOI":"10.24963\/ijcai.2018\/548"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i1.16156"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403384"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219823"}],"event":{"name":"WWW '22: The ACM Web Conference 2022","location":"Virtual Event, Lyon France","acronym":"WWW '22","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM Web Conference 2022"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3485447.3512109","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3485447.3512109","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:30:08Z","timestamp":1750188608000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3485447.3512109"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,4,25]]},"references-count":26,"alternative-id":["10.1145\/3485447.3512109","10.1145\/3485447"],"URL":"https:\/\/doi.org\/10.1145\/3485447.3512109","relation":{},"subject":[],"published":{"date-parts":[[2022,4,25]]},"assertion":[{"value":"2022-04-25","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}