{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T01:43:14Z","timestamp":1787017394520,"version":"build-2736575974"},"publisher-location":"New York, NY, USA","reference-count":68,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62072450"],"award-info":[{"award-number":["62072450"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,9,22]]},"DOI":"10.1145\/3705328.3748046","type":"proceedings-article","created":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T10:48:44Z","timestamp":1757155724000},"page":"114-123","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Exploring Scaling Laws of CTR Model for Online Performance Improvement"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-3600-9618","authenticated-orcid":false,"given":"Weijiang","family":"Lai","sequence":"first","affiliation":[{"name":"Institute of Software,Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3683-4034","authenticated-orcid":false,"given":"Beihong","family":"Jin","sequence":"additional","affiliation":[{"name":"Institute of Software Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-5574-4030","authenticated-orcid":false,"given":"Jiongyan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Meituan, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-3356-8446","authenticated-orcid":false,"given":"Yiyuan","family":"Zheng","sequence":"additional","affiliation":[{"name":"Institute of Software Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8498-9589","authenticated-orcid":false,"given":"Jian","family":"Dong","sequence":"additional","affiliation":[{"name":"Meituan, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1702-4263","authenticated-orcid":false,"given":"Jia","family":"Cheng","sequence":"additional","affiliation":[{"name":"Meituan, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4015-8668","authenticated-orcid":false,"given":"Jun","family":"Lei","sequence":"additional","affiliation":[{"name":"Meituan, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5495-0827","authenticated-orcid":false,"given":"Xingxing","family":"Wang","sequence":"additional","affiliation":[{"name":"Meituan, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,9,7]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"Josh Achiam Steven Adler Sandhini Agarwal Lama Ahmad Ilge Akkaya Florencia\u00a0Leoni Aleman Diogo Almeida Janko Altenschmidt Sam Altman Shyamal Anadkat et\u00a0al. 2023. Gpt-4 technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.08774 (2023)."},{"key":"e_1_3_3_2_3_2","unstructured":"Newsha Ardalani Carole-Jean Wu Zeliang Chen Bhargav Bhushanam and Adnan Aziz. 2022. Understanding scaling laws for recommendation models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2208.08489 (2022)."},{"key":"e_1_3_3_2_4_2","unstructured":"Jinze Bai Shuai Bai Yunfei Chu Zeyu Cui Kai Dang Xiaodong Deng Yang Fan Wenbin Ge Yu Han Fei Huang et\u00a0al. 2023. Qwen technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.16609 (2023)."},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","DOI":"10.1145\/3488560.3498435"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3495883"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"publisher","DOI":"10.1145\/1150402.1150464"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3511808.3557082"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599922"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5746"},{"key":"e_1_3_3_2_11_2","unstructured":"Qiwei Chen Changhua Pei Shanshan Lv Chao Li Junfeng Ge and Wenwu Ou. 2021. End-to-end user behavior retrieval in click-through rate prediction model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2108.04468 (2021)."},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/3326937.3341261"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/2988450.2988454"},{"key":"e_1_3_3_2_14_2","volume-title":"KDD 2023 Workshop on Artificial Intelligence for Computational Advertising (AdKDD)","author":"Chitlangia Sharad","year":"2023","unstructured":"Sharad Chitlangia, Krishna\u00a0Reddy Kesari, and Rajat Agarwal. 2023. Scaling generative pre-training for user ad activity sequences. In KDD 2023 Workshop on Artificial Intelligence for Computational Advertising (AdKDD). https:\/\/www.amazon.science\/publications\/scaling-generative-pre-training-for-user-ad-activity-sequences"},{"key":"e_1_3_3_2_15_2","first-page":"4057","volume-title":"International conference on machine learning","author":"Clark Aidan","year":"2022","unstructured":"Aidan Clark, Diego de Las\u00a0Casas, Aurelia Guy, Arthur Mensch, Michela Paganini, Jordan Hoffmann, Bogdan Damoc, Blake Hechtman, Trevor Cai, Sebastian Borgeaud, et\u00a0al. 2022. Unified scaling laws for routed language models. In International conference on machine learning. PMLR, 4057\u20134086."},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/2959100.2959190"},{"key":"e_1_3_3_2_17_2","unstructured":"Tri Dao Dan Fu Stefano Ermon Atri Rudra and Christopher R\u00e9. 2022. Flashattention: Fast and memory-efficient exact attention with io-awareness. Advances in Neural Information Processing Systems 35 (2022) 16344\u201316359."},{"key":"e_1_3_3_2_18_2","first-page":"7480","volume-title":"International Conference on Machine Learning","author":"Dehghani Mostafa","year":"2023","unstructured":"Mostafa Dehghani, Josip Djolonga, Basil Mustafa, Piotr Padlewski, Jonathan Heek, Justin Gilmer, Andreas\u00a0Peter Steiner, Mathilde Caron, Robert Geirhos, Ibrahim Alabdulmohsin, et\u00a0al. 2023. Scaling vision transformers to 22 billion parameters. In International Conference on Machine Learning. PMLR, 7480\u20137512."},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3543873.3584662"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657743"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.5555\/3367243.3367359"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","DOI":"10.1145\/3589335.3648334"},{"key":"e_1_3_3_2_23_2","volume-title":"The Eleventh International Conference on Learning Representations","author":"Frantar Elias","year":"2022","unstructured":"Elias Frantar, Saleh Ashkboos, Torsten Hoefler, and Dan Alistarh. 2022. OPTQ: Accurate quantization for generative pre-trained transformers. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"crossref","unstructured":"Alex Graves and Alex Graves. 2012. Long short-term memory. Supervised sequence labelling with recurrent neural networks (2012) 37\u201345.","DOI":"10.1007\/978-3-642-24797-2_4"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/239"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","DOI":"10.5555\/3692070.3692741"},{"key":"e_1_3_3_2_27_2","unstructured":"Tom Henighan Jared Kaplan Mor Katz Mark Chen Christopher Hesse Jacob Jackson Heewoo Jun Tom\u00a0B Brown Prafulla Dhariwal Scott Gray et\u00a0al. 2020. Scaling laws for autoregressive generative modeling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2010.14701 (2020)."},{"key":"e_1_3_3_2_28_2","unstructured":"Geoffrey Hinton Oriol Vinyals and Jeff Dean. 2015. Distilling the knowledge in a neural network. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1503.02531 (2015)."},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.5555\/3600270.3602446"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00745"},{"key":"e_1_3_3_2_31_2","first-page":"4904","volume-title":"International conference on machine learning","author":"Jia Chao","year":"2021","unstructured":"Chao Jia, Yinfei Yang, Ye Xia, Yi-Ting Chen, Zarana Parekh, Hieu Pham, Quoc Le, Yun-Hsuan Sung, Zhen Li, and Tom Duerig. 2021. Scaling up visual and vision-language representation learning with noisy text supervision. In International conference on machine learning. PMLR, 4904\u20134916."},{"key":"e_1_3_3_2_32_2","unstructured":"Jared Kaplan Sam McCandlish Tom Henighan Tom\u00a0B Brown Benjamin Chess Rewon Child Scott Gray Alec Radford Jeffrey Wu and Dario Amodei. 2020. Scaling laws for neural language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2001.08361 (2020)."},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1145\/3640457.3688055"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00060"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220023"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"crossref","unstructured":"Jianghao Lin Xinyi Dai Yunjia Xi Weiwen Liu Bo Chen Hao Zhang Yong Liu Chuhan Wu Xiangyang Li Chenxu Zhu et\u00a0al. 2025. How can recommender systems benefit from large language models: A survey. ACM Transactions on Information Systems 43 2 (2025) 1\u201347.","DOI":"10.1145\/3678004"},{"key":"e_1_3_3_2_37_2","volume-title":"Forty-first International Conference on Machine Learning","author":"Ludziejewski Jan","year":"2024","unstructured":"Jan Ludziejewski, Jakub Krajewski, Kamil Adamczewski, Maciej Pi\u00f3ro, Micha\u0142 Krutul, Szymon Antoniak, Kamil Ciebiera, Krystian Kr\u00f3l, Tomasz Odrzyg\u00f3\u017ad\u017a, Piotr Sankowski, et\u00a0al. 2024. Scaling laws for fine-grained mixture of experts. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.1145\/3604915.3608793"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220007"},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3210104"},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"crossref","unstructured":"Hieu Pham Zihang Dai Golnaz Ghiasi Kenji Kawaguchi Hanxiao Liu Adams\u00a0Wei Yu Jiahui Yu Yi-Ting Chen Minh-Thang Luong Yonghui Wu et\u00a0al. 2023. Combined scaling for zero-shot transfer learning. Neurocomputing 555 (2023) 126658.","DOI":"10.1016\/j.neucom.2023.126658"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330666"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1145\/3340531.3412744"},{"key":"e_1_3_3_2_44_2","unstructured":"Reiner Pope Sholto Douglas Aakanksha Chowdhery Jacob Devlin James Bradbury Jonathan Heek Kefan Xiao Shivani Agrawal and Jeff Dean. 2023. Efficiently scaling transformer inference. Proceedings of Machine Learning and Systems 5 (2023) 606\u2013624."},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401440"},{"key":"e_1_3_3_2_46_2","unstructured":"Colin Raffel Noam Shazeer Adam Roberts Katherine Lee Sharan Narang Michael Matena Yanqi Zhou Wei Li and Peter\u00a0J Liu. 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. Journal of machine learning research 21 140 (2020) 1\u201367."},{"key":"e_1_3_3_2_47_2","unstructured":"Prajit Ramachandran Barret Zoph and Quoc\u00a0V Le. 2017. Searching for activation functions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1710.05941 (2017)."},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"publisher","DOI":"10.1145\/3331184.3331230"},{"key":"e_1_3_3_2_49_2","unstructured":"Noam Shazeer. 2020. GLU variants improve transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2002.05202 (2020)."},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i4.25582"},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"publisher","DOI":"10.1145\/3627673.3680030"},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"publisher","unstructured":"Yi Tay Mostafa Dehghani Dara Bahri and Donald Metzler. 2022. Efficient Transformers: A Survey. 55 6 Article 109 (Dec. 2022) 28\u00a0pages. 10.1145\/3530811","DOI":"10.1145\/3530811"},{"key":"e_1_3_3_2_53_2","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard Xavier Martinet Marie-Anne Lachaux Timoth\u00e9e Lacroix Baptiste Rozi\u00e8re Naman Goyal Eric Hambro Faisal Azhar et\u00a0al. 2023. LLaMA: Open and efficient foundation language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.13971 (2023)."},{"key":"e_1_3_3_2_54_2","unstructured":"A Vaswani. 2017. Attention is all you need. Advances in Neural Information Processing Systems (2017)."},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3124749.3124754"},{"key":"e_1_3_3_2_56_2","doi-asserted-by":"publisher","DOI":"10.1145\/3442381.3450078"},{"key":"e_1_3_3_2_57_2","first-page":"802","volume-title":"International conference on machine learning","author":"Yan Ling","year":"2014","unstructured":"Ling Yan, Wu-Jun Li, Gui-Rong Xue, and Dingyi Han. 2014. Coupled group lasso for web-scale ctr prediction in display advertising. In International conference on machine learning. PMLR, 802\u2013810."},{"key":"e_1_3_3_2_58_2","first-page":"58484","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"Zhai Jiaqi","year":"2024","unstructured":"Jiaqi Zhai, Lucy Liao, Xing Liu, Yueming Wang, Rui Li, Xuan Cao, Leon Gao, Zhaojie Gong, Fangda Gu, Jiayuan He, et\u00a0al. 2024. Actions speak louder than words: trillion-parameter sequential transducers for generative recommendations. In Proceedings of the 41st International Conference on Machine Learning. 58484\u201358509."},{"key":"e_1_3_3_2_59_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01179"},{"key":"e_1_3_3_2_60_2","first-page":"59421","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"Zhang Buyun","year":"2024","unstructured":"Buyun Zhang, Liang Luo, Yuxin Chen, Jade Nie, Xi Liu, Shen Li, Yanli Zhao, Yuchen Hao, Yantao Yao, Ellie\u00a0Dingqiao Wen, et\u00a0al. 2024. Wukong: towards a scaling law for large-scale recommendation. In Proceedings of the 41st International Conference on Machine Learning. 59421\u201359434."},{"key":"e_1_3_3_2_61_2","unstructured":"Biao Zhang and Rico Sennrich. 2019. Root mean square layer normalization. Advances in Neural Information Processing Systems 32 (2019)."},{"key":"e_1_3_3_2_62_2","doi-asserted-by":"publisher","DOI":"10.1145\/3640457.3688129"},{"key":"e_1_3_3_2_63_2","unstructured":"Mingyang Zhang Hao Chen Chunhua Shen Zhen Yang Linlin Ou Xinyi Yu and Bohan Zhuang. 2023. LoRAPrune: Pruning meets low-rank parameter-efficient fine-tuning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.18403 (2023)."},{"key":"e_1_3_3_2_64_2","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3614963"},{"key":"e_1_3_3_2_65_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00454"},{"key":"e_1_3_3_2_66_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33015941"},{"key":"e_1_3_3_2_67_2","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219823"},{"key":"e_1_3_3_2_68_2","unstructured":"Xunyu Zhu Jian Li Yong Liu Can Ma and Weiping Wang. 2023. A survey on model compression for large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.07633 (2023)."},{"key":"e_1_3_3_2_69_2","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657816"}],"event":{"name":"RecSys '25: Nineteenth ACM Conference on Recommender Systems","location":"Prague Czech Republic","acronym":"RecSys '25","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction","SIGAI ACM Special Interest Group on Artificial Intelligence","SIGIR ACM Special Interest Group on Information Retrieval","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the Nineteenth ACM Conference on Recommender Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3705328.3748046","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T11:47:55Z","timestamp":1757159275000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3705328.3748046"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,7]]},"references-count":68,"alternative-id":["10.1145\/3705328.3748046","10.1145\/3705328"],"URL":"https:\/\/doi.org\/10.1145\/3705328.3748046","relation":{},"subject":[],"published":{"date-parts":[[2025,9,7]]},"assertion":[{"value":"2025-09-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}