{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T21:18:08Z","timestamp":1787692688201,"version":"build-2784847793"},"publisher-location":"New York, NY, USA","reference-count":110,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,8,9]]},"DOI":"10.1145\/3770854.3783952","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T12:07:40Z","timestamp":1785499660000},"page":"2335-2346","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Meta Lattice: Model Space Redesign for Cost-Effective Industry-Scale Ads Recommendations"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-6781-0693","authenticated-orcid":false,"given":"Liang","family":"Luo","sequence":"first","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-3320-0159","authenticated-orcid":false,"given":"Yuxin","family":"Chen","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9287-9756","authenticated-orcid":false,"given":"Zhengyu","family":"Zhang","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4125-2135","authenticated-orcid":false,"given":"Mengyue","family":"Hang","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1236-9484","authenticated-orcid":false,"given":"Andrew","family":"Gu","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-9053-4661","authenticated-orcid":false,"given":"Buyun","family":"Zhang","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2540-0712","authenticated-orcid":false,"given":"Boyang","family":"Liu","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-9359-3689","authenticated-orcid":false,"given":"Chen","family":"Chen","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9714-2465","authenticated-orcid":false,"given":"Fan","family":"Yang","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-0806-5672","authenticated-orcid":false,"given":"Feifan","family":"Gu","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2659-0833","authenticated-orcid":false,"given":"Huayu","family":"Li","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1664-2805","authenticated-orcid":false,"given":"Jade","family":"Nie","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9091-6412","authenticated-orcid":false,"given":"Jiayi","family":"Xu","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-5946-5456","authenticated-orcid":false,"given":"Jiyan","family":"Yang","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4750-9440","authenticated-orcid":false,"given":"Jongsoo","family":"Park","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6384-3810","authenticated-orcid":false,"given":"Laming","family":"Chen","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8909-4183","authenticated-orcid":false,"given":"Longhao","family":"Jin","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3031-0208","authenticated-orcid":false,"given":"Qin","family":"Huang","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-7999-6029","authenticated-orcid":false,"given":"Shali","family":"Jiang","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6373-7398","authenticated-orcid":false,"given":"Shiwen","family":"Shen","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-0182-4187","authenticated-orcid":false,"given":"Shuaiwen","family":"Wang","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4128-742X","authenticated-orcid":false,"given":"Siyang","family":"Yuan","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1223-5922","authenticated-orcid":false,"given":"Tongyi","family":"Tang","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9380-1010","authenticated-orcid":false,"given":"Weilin","family":"Zhang","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2336-8417","authenticated-orcid":false,"given":"Xi","family":"Liu","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-5751-3830","authenticated-orcid":false,"given":"Xiaohan","family":"Wei","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-8513-9566","authenticated-orcid":false,"given":"Yuchen","family":"Hao","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-5069-6298","authenticated-orcid":false,"given":"Xiaozhen","family":"Xia","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8598-2554","authenticated-orcid":false,"given":"Yasmine","family":"Badr","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3756-9949","authenticated-orcid":false,"given":"Zeliang","family":"Chen","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1803-6477","authenticated-orcid":false,"given":"Chengze","family":"Fan","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1832-0765","authenticated-orcid":false,"given":"Dong","family":"Liang","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2566-4316","authenticated-orcid":false,"given":"Qianru","family":"Li","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4302-1701","authenticated-orcid":false,"given":"Sihan","family":"Zeng","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-6451-809X","authenticated-orcid":false,"given":"Wenjun","family":"Wang","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7191-1549","authenticated-orcid":false,"given":"Yunlong","family":"He","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7036-2590","authenticated-orcid":false,"given":"Yinbin","family":"Ma","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6102-2903","authenticated-orcid":false,"given":"Maxim","family":"Naumov","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-3838-2317","authenticated-orcid":false,"given":"Yantao","family":"Yao","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-2843-9290","authenticated-orcid":false,"given":"Wenlin","family":"Chen","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8229-2294","authenticated-orcid":false,"given":"Ellie Dingqiao","family":"Wen","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,20]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"[n.d.]. Ads Quality. https:\/\/www.facebook.com\/business\/help\/423781975167984. [Accessed 11-07-2025]."},{"key":"e_1_3_2_2_2_1","unstructured":"[n.d.]. dual annealing scipy manual \u2014 docs.scipy.org. https:\/\/docs.scipy.org\/doc\/scipy\/reference\/generated\/scipy.optimize.dual_annealing.html. [Accessed 05-07-2025]."},{"key":"e_1_3_2_2_3_1","unstructured":"[n.d.]. Our next generation Meta Training and Inference Accelerator. https:\/\/ai.meta.com\/blog\/next-generation-meta-training-inference-accelerator-AI-MTIA\/. (Accessed on 05\/29\/2024)."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00653"},{"key":"e_1_3_2_2_5_1","volume-title":"Divya Mahajan, and Prashant J. Nair.","author":"Adnan Muhammad","year":"2023","unstructured":"Muhammad Adnan, Yassaman Ebrahimzadeh Maboud, Divya Mahajan, and Prashant J. Nair. 2023. Ad-Rec: Advanced Feature Interactions to Address Covariate-Shifts in Recommendation Networks. arXiv:2308.14902 [cs.IR]"},{"key":"e_1_3_2_2_6_1","volume-title":"Understanding scaling laws for recommendation models. arXiv preprint arXiv:2208.08489","author":"Ardalani Newsha","year":"2022","unstructured":"Newsha Ardalani, Carole-Jean Wu, Zeliang Chen, Bhargav Bhushanam, and Adnan Aziz. 2022. Understanding scaling laws for recommendation models. arXiv preprint arXiv:2208.08489 (2022)."},{"key":"e_1_3_2_2_7_1","volume-title":"Jamie Ryan Kiros, and Geoffrey E. Hinton","author":"Ba Jimmy Lei","year":"2016","unstructured":"Jimmy Lei Ba, Jamie Ryan Kiros, and Geoffrey E. Hinton. 2016. Layer Normalization. arXiv:1607.06450 [stat.ML] https:\/\/arxiv.org\/abs\/1607.06450"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1613\/jair.731"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.25080\/Majora-8b375195-004"},{"key":"e_1_3_2_2_10_1","unstructured":"Rishi Bommasani Drew A Hudson Ehsan Adeli Russ Altman Simran Arora Sydney von Arx Michael S Bernstein Jeannette Bohg Antoine Bosselut Emma Brunskill et al. 2021. On the opportunities and risks of foundation models. arXiv preprint arXiv:2108.07258 (2021)."},{"key":"e_1_3_2_2_11_1","volume-title":"Random forests. Machine learning 45","author":"Breiman Leo","year":"2001","unstructured":"Leo Breiman. 2001. Random forests. Machine learning 45 (2001), 5-32."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599884"},{"key":"e_1_3_2_2_13_1","unstructured":"Qiwei Chen Huan Zhao Wei Li Pipei Huang and Wenwu Ou. 2019. Behavior Sequence Transformer for E-commerce Recommendation in Alibaba. arXiv:1905.06874 [cs.IR] https:\/\/arxiv.org\/abs\/1905.06874"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3511965"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5768"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/1367497.1367646"},{"key":"e_1_3_2_2_17_1","volume-title":"International Conference on Machine Learning. PMLR, 7480-7512","author":"Dehghani Mostafa","year":"2023","unstructured":"Mostafa Dehghani, Josip Djolonga, Basil Mustafa, Piotr Padlewski, Jonathan Heek, Justin Gilmer, Andreas Peter Steiner, Mathilde Caron, Robert Geirhos, Ibrahim Alabdulmohsin, et al. 2023. Scaling vision transformers to 22 billion parameters. In International Conference on Machine Learning. PMLR, 7480-7512."},{"key":"e_1_3_2_2_18_1","volume-title":"Emerging Properties in Unified Multimodal Pretraining. arXiv preprint arXiv:2505.14683","author":"Deng Chaorui","year":"2025","unstructured":"Chaorui Deng, Deyao Zhu, Kunchang Li, Chenhui Gou, Feng Li, Zeyu Wang, Shu Zhong, Weihao Yu, Xiaonan Nie, Ziang Song, Guang Shi, and Haoqi Fan. 2025. Emerging Properties in Unified Multimodal Pretraining. arXiv preprint arXiv:2505.14683 (2025)."},{"key":"e_1_3_2_2_19_1","first-page":"27503","article-title":"Efficiently identifying task groupings for multi-task learning","volume":"34","author":"Fifty Chris","year":"2021","unstructured":"Chris Fifty, Ehsan Amid, Zhe Zhao, Tianhe Yu, Rohan Anil, and Chelsea Finn. 2021. Efficiently identifying task groupings for multi-task learning. Advances in Neural Information Processing Systems 34 (2021), 27503-27516.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_20_1","volume-title":"Vip5: Towards multimodal foundation models for recommendation. arXiv preprint arXiv:2305.14302","author":"Geng Shijie","year":"2023","unstructured":"Shijie Geng, Juntao Tan, Shuchang Liu, Zuohui Fu, and Yongfeng Zhang. 2023. Vip5: Towards multimodal foundation models for recommendation. arXiv preprint arXiv:2305.14302 (2023)."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"crossref","unstructured":"Dirk Groeneveld Iz Beltagy Pete Walsh Akshita Bhagia Rodney Kinney Oyvind Tafjord Ananya Harsh Jha Hamish Ivison Ian Magnusson Yizhong Wang Shane Arora David Atkinson Russell Authur Khyathi Raghavi Chandu Arman Cohan Jennifer Dumas Yanai Elazar Yuling Gu Jack Hessel Tushar Khot William Merrill Jacob Morrison Niklas Muennighoff Aakanksha Naik Crystal Nam Matthew E. Peters Valentina Pyatkin Abhilasha Ravichander Dustin Schwenk Saurabh Shah Will Smith Emma Strubell Nishant Subramani Mitchell Wortsman Pradeep Dasigi Nathan Lambert Kyle Richardson Luke Zettlemoyer Jesse Dodge Kyle Lo Luca Soldaini Noah A. Smith and Hannaneh Hajishirzi. 2024. OLMo: Accelerating the Science of Language Models. arXiv:2402.00838 [cs.CL] https:\/\/arxiv.org\/abs\/2402.00838","DOI":"10.18653\/v1\/2024.acl-long.841"},{"key":"e_1_3_2_2_22_1","volume-title":"On the Embedding Collapse when Scaling up Recommendation Models. arXiv preprint arXiv:2310.04400","author":"Guo Xingzhuo","year":"2023","unstructured":"Xingzhuo Guo, Junwei Pan, Ximei Wang, Baixu Chen, Jie Jiang, and Mingsheng Long. 2023. On the Embedding Collapse when Scaling up Recommendation Models. arXiv preprint arXiv:2310.04400 (2023)."},{"key":"e_1_3_2_2_23_1","volume-title":"Shampoo: Preconditioned Stochastic Tensor Optimization. arXiv:1802.09568 [cs.LG] https:\/\/arxiv.org\/abs\/1802.09568","author":"Gupta Vineet","year":"2018","unstructured":"Vineet Gupta, Tomer Koren, and Yoram Singer. 2018. Shampoo: Preconditioned Stochastic Tensor Optimization. arXiv:1802.09568 [cs.LG] https:\/\/arxiv.org\/abs\/1802.09568"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3512093"},{"key":"e_1_3_2_2_25_1","volume-title":"Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531","author":"Hinton Geoffrey","year":"2015","unstructured":"Geoffrey Hinton, Oriol Vinyals, and Jeff Dean. 2015. Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531 (2015)."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"crossref","unstructured":"Jordan Hoffmann Sebastian Borgeaud Arthur Mensch Elena Buchatskaya Trevor Cai Eliza Rutherford Diego de Las Casas Lisa Anne Hendricks Johannes Welbl Aidan Clark Tom Hennigan Eric Noland Katie Millican George van den Driessche Bogdan Damoc Aurelia Guy Simon Osindero Karen Simonyan Erich Elsen Jack W. Rae Oriol Vinyals and Laurent Sifre. 2022. Training Compute-Optimal Large Language Models. arXiv:2203.15556 [cs.CL] https:\/\/arxiv.org\/abs\/2203.15556","DOI":"10.52202\/068431-2176"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"crossref","unstructured":"Andrew Howard Mark Sandler Grace Chu Liang-Chieh Chen Bo Chen Mingxing Tan WeijunWang Yukun Zhu Ruoming Pang Vijay Vasudevan Quoc V. Le and Hartwig Adam. 2019. Searching for MobileNetV3. arXiv:1905.02244 [cs.CV] https:\/\/arxiv.org\/abs\/1905.02244","DOI":"10.1109\/ICCV.2019.00140"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3523227.3547387"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2939756"},{"key":"e_1_3_2_2_31_1","volume-title":"Scaling laws for neural language models. arXiv preprint arXiv:2001.08361","author":"Kaplan Jared","year":"2020","unstructured":"Jared Kaplan, Sam McCandlish, Tom Henighan, Tom B Brown, Benjamin Chess, Rewon Child, Scott Gray, Alec Radford, Jeffrey Wu, and Dario Amodei. 2020. Scaling laws for neural language models. arXiv preprint arXiv:2001.08361 (2020)."},{"key":"e_1_3_2_2_32_1","volume-title":"FBGEMM: Enabling High-Performance Low-Precision Deep Learning Inference. arXiv:2101.05615 [cs.LG] https:\/\/arxiv.org\/abs\/2101.05615","author":"Khudia Daya","year":"2021","unstructured":"Daya Khudia, Jianyu Huang, Protonu Basu, Summer Deng, Haixin Liu, Jongsoo Park, and Mikhail Smelyanskiy. 2021. FBGEMM: Enabling High-Performance Low-Precision Deep Learning Inference. arXiv:2101.05615 [cs.LG] https:\/\/arxiv.org\/abs\/2101.05615"},{"key":"e_1_3_2_2_33_1","volume-title":"C Daniel Gelatt Jr, and Mario P Vecchi","author":"Kirkpatrick Scott","year":"1983","unstructured":"Scott Kirkpatrick, C Daniel Gelatt Jr, and Mario P Vecchi. 1983. Optimization by simulated annealing. science 220, 4598 (1983), 671-680."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3298689.3347002"},{"key":"e_1_3_2_2_35_1","unstructured":"Kuaishou. [n.d.]. https:\/\/www.kuaishou.com\/activity\/uimc"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3604915.3608828"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599769"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3336191.3371793"},{"key":"e_1_3_2_2_39_1","unstructured":"Shen Li Yanli Zhao Rohan Varma Omkar Salpekar Pieter Noordhuis Teng Li Adam Paszke Jeff Smith Brian Vaughan Pritam Damania and Soumith Chintala. 2020. PyTorch Distributed: Experiences on Accelerating Data Parallel Training. arXiv:2006.15704 [cs.DC] https:\/\/arxiv.org\/abs\/2006.15704"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220023"},{"key":"e_1_3_2_2_41_1","volume-title":"Persia: An open, hybrid system scaling deep learning-based recommenders up to 100 trillion parameters. (Nov.","author":"Lian Xiangru","year":"2021","unstructured":"Xiangru Lian, Binhang Yuan, Xuefeng Zhu, Yulong Wang, Yongjun He, Honghuan Wu, Lei Sun, Haodong Lyu, Chengjun Liu, Xing Dong, Yiqiao Liao, Mingnan Luo, Congfei Zhang, Jingru Xie, Haonan Li, Lei Chen, Renjie Huang, Jianying Lin, Chengchun Shu, Xuezhong Qiu, Zhishan Liu, Dongying Kong, Lei Yuan, Hai Yu, Sen Yang, Ce Zhang, and Ji Liu. 2021. Persia: An open, hybrid system scaling deep learning-based recommenders up to 100 trillion parameters. (Nov. 2021). arXiv:2111.05897 [cs.LG]"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","unstructured":"Mingfu Liang Xi Liu Rong Jin Boyang Liu Qiuling Suo Qinghai Zhou Song Zhou Laming Chen Hua Zheng Zhiyuan Li Shali Jiang Jiyan Yang Xiaozhen Xia Fan Yang Yasmine Badr Ellie Wen Shuyu Xu Hansey Chen Zhengyu Zhang and Huayu Li. 2025. External Large Foundation Model: Howto Efficiently Serve Trillions of Parameters for Online Ads Recommendation. doi:10.48550\/arXiv.2502.17494","DOI":"10.48550\/arXiv.2502.17494"},{"key":"e_1_3_2_2_43_1","unstructured":"Wanchao Liang Tianyu Liu Less Wright Will Constable Andrew Gu Chien-Chin Huang Iris Zhang Wei Feng Howard Huang Junjie Wang Sanket Purandare Gokul Nadathur and Stratos Idreos. 2024. TorchTitan: One-stop PyTorch native solution for production ready LLM pre-training. arXiv:2410.06511 [cs.CL] https:\/\/arxiv.org\/abs\/2410.06511"},{"key":"e_1_3_2_2_44_1","volume-title":"Luke Zettlemoyer, and Xi Victoria Lin.","author":"Liang Weixin","year":"2024","unstructured":"Weixin Liang, Lili Yu, Liang Luo, Srinivasan Iyer, Ning Dong, Chunting Zhou, Gargi Ghosh, Mike Lewis, Wen tau Yih, Luke Zettlemoyer, and Xi Victoria Lin. 2024. Mixture-of-Transformers: A Sparse and Scalable Architecture for Multi-Modal Foundation Models. arXiv:2411.04996 [cs.CL] https:\/\/arxiv.org\/abs\/2411.04996"},{"key":"e_1_3_2_2_45_1","volume-title":"andWeilin Huang","author":"Liao Chao","year":"2025","unstructured":"Chao Liao, Liyang Liu, Xun Wang, Zhengxiong Luo, Xinyu Zhang, Wenliang Zhao, JieWu, Liang Li, Zhi Tian, andWeilin Huang. 2025. Mogao: An Omni Foundation Model for Interleaved Multi-Modal Generation. arXiv:2505.05472 [cs.CV] https:\/\/arxiv.org\/abs\/2505.05472"},{"key":"e_1_3_2_2_46_1","unstructured":"Xi Victoria Lin Akshat Shrivastava Liang Luo Srinivasan Iyer Mike Lewis Gargi Gosh Luke Zettlemoyer and Armen Aghajanyan. 2024. MoMa: Efficient Early-Fusion Pre-training with Mixture of Modality-Aware Experts. arXiv:2407.21770 [cs.AI] https:\/\/arxiv.org\/abs\/2407.21770"},{"key":"e_1_3_2_2_47_1","volume-title":"Mamba4rec: Towards efficient sequential recommendation with selective state space models. arXiv preprint arXiv:2403.03900","author":"Liu Chengkai","year":"2024","unstructured":"Chengkai Liu, Jianghao Lin, Jianling Wang, Hanzhou Liu, and James Caverlee. 2024. Mamba4rec: Towards efficient sequential recommendation with selective state space models. arXiv preprint arXiv:2403.03900 (2024)."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3511808.3557140"},{"key":"e_1_3_2_2_49_1","unstructured":"Qidong Liu Jiaxi Hu Yutian Xiao Xiangyu Zhao Jingtong Gao Wanyu Wang Qing Li and Jiliang Tang. 2024. Multimodal Recommender Systems: A Survey. arXiv:2302.03883 [cs.IR] https:\/\/arxiv.org\/abs\/2302.03883"},{"key":"e_1_3_2_2_50_1","volume-title":"Multi-epoch learning for deep Click-Through Rate prediction models. arXiv [cs.IR] (May","author":"Liu Zhaocheng","year":"2023","unstructured":"Zhaocheng Liu, Zhongxiang Fan, Jian Liang, Dongying Kong, and Han Li. 2023. Multi-epoch learning for deep Click-Through Rate prediction models. arXiv [cs.IR] (May 2023)."},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3267809.3267840"},{"key":"e_1_3_2_2_52_1","first-page":"82","article-title":"Plink: Discovering and exploiting locality for accelerated distributed training on the public cloud","volume":"2","author":"Luo Liang","year":"2020","unstructured":"Liang Luo, Peter West, Jacob Nelson, Arvind Krishnamurthy, and Luis Ceze. 2020. Plink: Discovering and exploiting locality for accelerated distributed training on the public cloud. Proceedings of Machine Learning and Systems 2 (2020), 82-97.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_2_53_1","unstructured":"Liang Luo Buyun Zhang Michael Tsang Yinbin Ma Ching-Hsiang Chu Yuxin Chen Shen Li Yuchen Hao Yanli Zhao Guna Lakshminarayanan et al. 2024. Disaggregated Multi-Tower: Topology-aware Modeling Technique for Efficient Large-Scale Recommendation. arXiv preprint arXiv:2403.00877 (2024)."},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220007"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3210104"},{"key":"e_1_3_2_2_56_1","volume-title":"Dropped scheduled task: Mitigating negative transfer in multi-task learning using dynamic task dropping. Transactions on Machine Learning Research","author":"Malhotra Aakarsh","year":"2022","unstructured":"Aakarsh Malhotra, Mayank Vatsa, and Richa Singh. 2022. Dropped scheduled task: Mitigating negative transfer in multi-task learning using dynamic task dropping. Transactions on Machine Learning Research (2022)."},{"key":"e_1_3_2_2_57_1","volume-title":"FinalMLP: An Enhanced Two-Stream MLP Model for CTR Prediction. arXiv preprint arXiv:2304.00902","author":"Mao Kelong","year":"2023","unstructured":"Kelong Mao, Jieming Zhu, Liangcai Su, Guohao Cai, Yuru Li, and Zhenhua Dong. 2023. FinalMLP: An Enhanced Two-Stream MLP Model for CTR Prediction. arXiv preprint arXiv:2304.00902 (2023)."},{"key":"e_1_3_2_2_58_1","volume-title":"Bounds for Linear Multi-Task Learning. J. Mach. Learn. Res. 7 (Dec","author":"Maurer Andreas","year":"2006","unstructured":"Andreas Maurer. 2006. Bounds for Linear Multi-Task Learning. J. Mach. Learn. Res. 7 (Dec. 2006), 117-139."},{"key":"e_1_3_2_2_59_1","unstructured":"Meta AI. 2023. AI Ads: Performance and Efficiency with Meta Lattice. https:\/\/ai.meta.com\/blog\/ai-ads-performance-efficiency-meta-lattice\/"},{"key":"e_1_3_2_2_60_1","unstructured":"Meta Production Engineering. 2024. The Andromeda Advantage: How Meta's Next-Gen Retrieval Engine Automates Personalized Ads. https:\/\/engineering.fb.com\/2024\/12\/02\/production-engineering\/meta-andromeda-advantage-automation-next-gen-personalized-ads-retrieval-engine\/"},{"key":"e_1_3_2_2_61_1","volume-title":"Revisiting multi-task learning with rock: a deep residual auxiliary block for visual detection. Advances in neural information processing systems 31","author":"Mordan Taylor","year":"2018","unstructured":"Taylor Mordan, Nicolas Thome, Gilles Henaff, and Matthieu Cord. 2018. Revisiting multi-task learning with rock: a deep residual auxiliary block for visual detection. Advances in neural information processing systems 31 (2018)."},{"key":"e_1_3_2_2_62_1","unstructured":"Dheevatsa Mudigere Yuchen Hao Jianyu Huang Andrew Tulloch Srinivas Sridharan Xing Liu Mustafa Ozdal Jade Nie Jongsoo Park Liang Luo et al. 2021. High-performance distributed training of large-scale deep learning recommendation models. arXiv preprint arXiv:2104.05158 (2021)."},{"key":"e_1_3_2_2_63_1","volume-title":"Adaptive smoothed online multi-task learning. Advances in Neural Information Processing Systems 29","author":"Murugesan Keerthiram","year":"2016","unstructured":"Keerthiram Murugesan, Hanxiao Liu, Jaime Carbonell, and Yiming Yang. 2016. Adaptive smoothed online multi-task learning. Advances in Neural Information Processing Systems 29 (2016)."},{"key":"e_1_3_2_2_64_1","unstructured":"Rafael M\u00fcller Simon Kornblith and Geoffrey Hinton. 2020. When Does Label Smoothing Help? arXiv:1906.02629 [cs.LG] https:\/\/arxiv.org\/abs\/1906.02629"},{"key":"e_1_3_2_2_65_1","volume-title":"Yi Tay, William Fedus, Thibault Fevry, Michael Matena, Karishma Malkan, Noah Fiedel, Noam Shazeer, Zhenzhong Lan, et al.","author":"Narang Sharan","year":"2021","unstructured":"Sharan Narang, Hyung Won Chung, Yi Tay, William Fedus, Thibault Fevry, Michael Matena, Karishma Malkan, Noah Fiedel, Noam Shazeer, Zhenzhong Lan, et al. 2021. Do transformer modifications transfer across implementations and applications? arXiv preprint arXiv:2102.11972 (2021)."},{"key":"e_1_3_2_2_66_1","volume-title":"Jianyu Huang, Narayanan Sundaraman, Jongsoo Park, Xiaodong Wang, Udit Gupta, Carole-Jean Wu, Alisson G Azzolini, et al.","author":"Naumov Maxim","year":"2019","unstructured":"Maxim Naumov, Dheevatsa Mudigere, Hao-Jun Michael Shi, Jianyu Huang, Narayanan Sundaraman, Jongsoo Park, Xiaodong Wang, Udit Gupta, Carole-Jean Wu, Alisson G Azzolini, et al. 2019. Deep learning recommendation model for personalization and recommendation systems. arXiv preprint arXiv:1906.00091 (2019)."},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2009.191"},{"key":"e_1_3_2_2_68_1","unstructured":"Jongsoo Park Maxim Naumov Protonu Basu Summer Deng Aravind Kalaiah Daya Shanker Khudia James Law Parth Malani Andrey Malevich Nadathur Satish Juan Miguel Pino Martin Schatz Alexander Sidorov Viswanath Sivakumar Andrew Tulloch Xiaodong Wang Yiming Wu Hector Yuen Utku Diril Dmytro Dzhulgakov Kim M. Hazelwood Bill Jia Yangqing Jia Lin Qiao Vijay Rao Nadav Rotem Sungjoo Yoo and Mikhail Smelyanskiy. 2018. Deep Learning Inference in Facebook Data Centers: Characterization Performance Optimizations and Hardware Implications. CoRR abs\/1811.09886 (2018). arXiv:1811.09886 http:\/\/arxiv.org\/abs\/1811.09886"},{"key":"e_1_3_2_2_69_1","volume-title":"International conference on machine learning. Pmlr, 1310-1318","author":"Pascanu Razvan","year":"2013","unstructured":"Razvan Pascanu, Tomas Mikolov, and Yoshua Bengio. 2013. On the difficulty of training recurrent neural networks. In International conference on machine learning. Pmlr, 1310-1318."},{"key":"e_1_3_2_2_70_1","volume-title":"International conference on machine learning. PMLR, 2807-2816","author":"Pentina Anastasia","year":"2017","unstructured":"Anastasia Pentina and Christoph H Lampert. 2017. Multi-task learning with labeled and unlabeled tasks. In International conference on machine learning. PMLR, 2807-2816."},{"key":"e_1_3_2_2_71_1","volume-title":"Le","author":"Ramachandran Prajit","year":"2017","unstructured":"Prajit Ramachandran, Barret Zoph, and Quoc V. Le. 2017. Swish: a Self-Gated Activation Function. arXiv: Neural and Evolutionary Computing (2017). https:\/\/api.semanticscholar.org\/CorpusID:196158220"},{"key":"e_1_3_2_2_72_1","unstructured":"Colfax Research. 2025. DeepSeek-R1 and FP8 Mixed-Precision Training. https:\/\/research.colfax-intl.com\/deepseek-r1-and-fp8-mixed-precision-training\/ Discussion of FP8 accumulation strategies including tensor core vs CUDA core accumulation."},{"key":"e_1_3_2_2_73_1","volume-title":"How does batch normalization help optimization? Advances in neural information processing systems 31","author":"Santurkar Shibani","year":"2018","unstructured":"Shibani Santurkar, Dimitris Tsipras, Andrew Ilyas, and Aleksander Madry. 2018. How does batch normalization help optimization? Advances in neural information processing systems 31 (2018)."},{"key":"e_1_3_2_2_74_1","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3481941"},{"key":"e_1_3_2_2_75_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i4.25582"},{"key":"e_1_3_2_2_76_1","volume-title":"Practical bayesian optimization of machine learning algorithms. Advances in neural information processing systems 25","author":"Snoek Jasper","year":"2012","unstructured":"Jasper Snoek, Hugo Larochelle, and Ryan P Adams. 2012. Practical bayesian optimization of machine learning algorithms. Advances in neural information processing systems 25 (2012)."},{"key":"e_1_3_2_2_77_1","doi-asserted-by":"publisher","DOI":"10.1145\/3357384.3357925"},{"key":"e_1_3_2_2_78_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1008202821328"},{"key":"e_1_3_2_2_79_1","unstructured":"Jianlin Su Yu Lu Shengfeng Pan Ahmed Murtadha Bo Wen and Yunfeng Liu. 2023. RoFormer: Enhanced Transformer with Rotary Position Embedding. arXiv:2104.09864 [cs.CL] https:\/\/arxiv.org\/abs\/2104.09864"},{"key":"e_1_3_2_2_80_1","doi-asserted-by":"publisher","DOI":"10.1145\/3383313.3412236"},{"key":"e_1_3_2_2_81_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599846"},{"key":"e_1_3_2_2_82_1","volume-title":"H Chi","author":"Tang Jiaxi","year":"2023","unstructured":"Jiaxi Tang, Yoel Drori, Daryl Chang, Maheswaran Sathiamoorthy, Justin Gilmer, Li Wei, Xinyang Yi, Lichan Hong, and Ed H Chi. 2023. Improving Training Stability for Multitask Ranking Models in Recommender Systems. arXiv preprint arXiv:2302.09178 (2023)."},{"key":"e_1_3_2_2_83_1","unstructured":"Linwei Tao Minjing Dong and Chang Xu. 2024. Feature Clipping for Uncertainty Calibration. arXiv:2410.19796 [cs.CV] https:\/\/arxiv.org\/abs\/2410.19796"},{"key":"e_1_3_2_2_84_1","volume-title":"Chameleon: Mixed-Modal Early-Fusion Foundation Models. arXiv:2405.09818 [cs.CL]","author":"Team Chameleon","year":"2024","unstructured":"Chameleon Team. 2024. Chameleon: Mixed-Modal Early-Fusion Foundation Models. arXiv:2405.09818 [cs.CL]"},{"key":"e_1_3_2_2_85_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_2_86_1","first-page":"1785","volume-title":"Proceedings of the web conference","author":"Shivanna Rakesh","year":"2021","unstructured":"RuoxiWang, Rakesh Shivanna, Derek Cheng, Sagar Jain, Dong Lin, Lichan Hong, and Ed Chi. 2021. Dcn v2: Improved deep & cross network and practical lessons for web-scale learning to rank systems. In Proceedings of the web conference 2021. 1785-1797."},{"key":"e_1_3_2_2_87_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539221"},{"key":"e_1_3_2_2_88_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467326"},{"key":"e_1_3_2_2_89_1","volume-title":"Delayed feed-back modeling for the entire space conversion rate prediction. arXiv preprint arXiv:2011.11826","author":"Wang Yanshi","year":"2020","unstructured":"Yanshi Wang, Jie Zhang, Qing Da, and Anxiang Zeng. 2020. Delayed feed-back modeling for the entire space conversion rate prediction. arXiv preprint arXiv:2011.11826 (2020)."},{"key":"e_1_3_2_2_90_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3512021"},{"key":"e_1_3_2_2_91_1","volume-title":"MaskNet: Introducing feature-wise multiplication to CTR ranking models by instance-guided mask. arXiv preprint arXiv:2102.07619","author":"Wang Zhiqiang","year":"2021","unstructured":"Zhiqiang Wang, Qingyun She, and Junlin Zhang. 2021. MaskNet: Introducing feature-wise multiplication to CTR ranking models by instance-guided mask. arXiv preprint arXiv:2102.07619 (2021)."},{"key":"e_1_3_2_2_92_1","volume-title":"Janus: Decoupling Visual Encoding for Unified Multimodal Understanding and Generation. arXiv:2410.13848 [cs.CV] https:\/\/arxiv.org\/abs\/2410.13848","author":"Wu Chengyue","year":"2024","unstructured":"Chengyue Wu, Xiaokang Chen, Zhiyu Wu, Yiyang Ma, Xingchao Liu, Zizheng Pan, Wen Liu, Zhenda Xie, Xingkai Yu, Chong Ruan, and Ping Luo. 2024. Janus: Decoupling Visual Encoding for Unified Multimodal Understanding and Generation. arXiv:2410.13848 [cs.CV] https:\/\/arxiv.org\/abs\/2410.13848"},{"key":"e_1_3_2_2_93_1","first-page":"795","article-title":"Sustainable ai: Environmental implications, challenges and opportunities","volume":"4","author":"Wu Carole-Jean","year":"2022","unstructured":"Carole-Jean Wu, Ramya Raghavendra, Udit Gupta, Bilge Acun, Newsha Ardalani, Kiwan Maeng, Gloria Chang, Fiona Aga, Jinshi Huang, Charles Bai, et al. 2022. Sustainable ai: Environmental implications, challenges and opportunities. Proceedings of Machine Learning and Systems 4 (2022), 795-813.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_2_94_1","unstructured":"Da Xiao Qingye Meng Shengping Li and Xingyuan Yuan. 2025. MUDDFormer: Breaking Residual Bottlenecks in Transformers via Multiway Dynamic Dense Connections. arXiv:2502.12170 [cs.LG] https:\/\/arxiv.org\/abs\/2502.12170"},{"key":"e_1_3_2_2_95_1","unstructured":"Jiayi Xin Sukwon Yun Jie Peng Inyoung Choi Jenna L. Ballard Tianlong Chen and Qi Long. 2025. I2MoE: Interpretable Multimodal Interaction-aware Mixture-of-Experts. arXiv:2505.19190 [cs.LG] https:\/\/arxiv.org\/abs\/2505.19190"},{"key":"e_1_3_2_2_96_1","first-page":"24740","volume-title":"Oh (Eds.)","volume":"35","author":"Yan Bencheng","year":"2022","unstructured":"Bencheng Yan, Pengjie Wang, Kai Zhang, Feng Li, Hongbo Deng, Jian Xu, and Bo Zheng. 2022. APG: Adaptive Parameter Generation Network for Click-Through Rate Prediction. In Advances in Neural Information Processing Systems, S. Koyejo, S. Mohamed, A. Agarwal, D. Belgrave, K. Cho, and A. Oh (Eds.), Vol. 35. Curran Associates, Inc., 24740-24752. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/9cd0c57170f48520749d5ae62838241f-Paper-Conference.pdf"},{"key":"e_1_3_2_2_97_1","first-page":"17","article-title":"Deepapf: Deep attentive probabilistic factorization for multi-site video recommendation","volume":"2","author":"Yan Huan","year":"2019","unstructured":"Huan Yan, Xiangning Chen, Chen Gao, Yong Li, and Depeng Jin. 2019. Deepapf: Deep attentive probabilistic factorization for multi-site video recommendation. TC 2, 130 (2019), 17-883.","journal-title":"TC"},{"key":"e_1_3_2_2_98_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i9.26275"},{"key":"e_1_3_2_2_99_1","doi-asserted-by":"publisher","DOI":"10.1145\/3511808.3557541"},{"key":"e_1_3_2_2_100_1","doi-asserted-by":"crossref","unstructured":"Zhichen Zeng Xiaolong Liu Mengyue Hang Xiaoyi Liu Qinghai Zhou Chaofei Yang Yiqun Liu Yichen Ruan Laming Chen Yuxin Chen Yujia Hao Jiaqi Xu Jade Nie Xi Liu Buyun Zhang Wei Wen Siyang Yuan Kai Wang Wen-Yen Chen Yiping Han Huayu Li Chunzhi Yang Bo Long Philip S. Yu Hanghang Tong and Jiyan Yang. 2024. InterFormer: Towards Effective Heterogeneous Interaction Learning for Click-Through Rate Prediction. arXiv:2411.09852 [cs.IR] https:\/\/arxiv.org\/abs\/2411.09852","DOI":"10.1145\/3746252.3761527"},{"key":"e_1_3_2_2_101_1","volume-title":"Proceedings of Machine Learning and Systems 5","author":"Zha Daochen","year":"2023","unstructured":"Daochen Zha, Louis Feng, Liang Luo, Bhargav Bhushanam, Zirui Liu, Yusuo Hu, Jade Nie, Yuzhen Huang, Yuandong Tian, Arun Kejariwal, et al. 2023. Pre-train and Search: Efficient Embedding Table Sharding with Pre-trained Neural Cost Models. Proceedings of Machine Learning and Systems 5 (2023)."},{"key":"e_1_3_2_2_102_1","volume-title":"Words: Trillion-Parameter Sequential Transducers for Generative Recommendations. arXiv preprint arXiv:2402.17152","author":"Zhai Jiaqi","year":"2024","unstructured":"Jiaqi Zhai, Lucy Liao, Xing Liu, Yueming Wang, Rui Li, Xuan Cao, Leon Gao, Zhaojie Gong, Fangda Gu, Michael He, et al. 2024. Actions Speak Louder than Words: Trillion-Parameter Sequential Transducers for Generative Recommendations. arXiv preprint arXiv:2402.17152 (2024)."},{"key":"e_1_3_2_2_103_1","volume-title":"Wukong: Towards a Scaling Law for Large-Scale Recommendation. arXiv preprint arXiv:2403.02545","author":"Zhang Buyun","year":"2024","unstructured":"Buyun Zhang, Liang Luo, Yuxin Chen, Jade Nie, Xi Liu, Daifeng Guo, Yanli Zhao, Shen Li, Yuchen Hao, Yantao Yao, et al. 2024. Wukong: Towards a Scaling Law for Large-Scale Recommendation. arXiv preprint arXiv:2403.02545 (2024)."},{"key":"e_1_3_2_2_104_1","volume-title":"DHEN: A deep and hierarchical ensemble network for large-scale click-through rate prediction. arXiv preprint arXiv:2203.11014","author":"Zhang Buyun","year":"2022","unstructured":"Buyun Zhang, Liang Luo, Xi Liu, Jay Li, Zeliang Chen, Weilin Zhang, Xiaohan Wei, Yuchen Hao, Michael Tsang, Wenjun Wang, et al. 2022. DHEN: A deep and hierarchical ensemble network for large-scale click-through rate prediction. arXiv preprint arXiv:2203.11014 (2022)."},{"key":"e_1_3_2_2_105_1","volume-title":"Wayne Xin Zhao, and Ji-Rong Wen","author":"Zhang Gaowei","year":"2023","unstructured":"Gaowei Zhang, Yupeng Hou, Hongyu Lu, Yu Chen, Wayne Xin Zhao, and Ji-Rong Wen. 2023. Scaling Law of Large Sequential Recommendation Models. arXiv preprint arXiv:2311.11351 (2023)."},{"key":"e_1_3_2_2_106_1","doi-asserted-by":"publisher","DOI":"10.1145\/3488560.3498479"},{"key":"e_1_3_2_2_107_1","unstructured":"Zijian Zhang Shuchang Liu Jiaao Yu Qingpeng Cai Xiangyu Zhao Chunxu Zhang Ziru Liu Qidong Liu Hongwei Zhao Lantao Hu et al. 2024. M3oE: Multi-Domain Multi-Task Mixture-of Experts Recommendation Framework. arXiv preprint arXiv:2404.18465 (2024)."},{"key":"e_1_3_2_2_108_1","doi-asserted-by":"crossref","unstructured":"Yanli Zhao Andrew Gu Rohan Varma Liang Luo Chien-Chin Huang Min Xu Less Wright Hamid Shojanazeri Myle Ott Sam Shleifer et al. 2023. Pytorch FSDP: experiences on scaling fully sharded data parallel. arXiv preprint arXiv:2304.11277 (2023).","DOI":"10.14778\/3611540.3611569"},{"key":"e_1_3_2_2_109_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3531723"},{"key":"e_1_3_2_2_110_1","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3482486"}],"event":{"name":"KDD '26: The 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Jeju Island Republic of Korea","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.1"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3770854.3783952","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T20:19:18Z","timestamp":1787689158000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3770854.3783952"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,20]]},"references-count":110,"alternative-id":["10.1145\/3770854.3783952","10.1145\/3770854"],"URL":"https:\/\/doi.org\/10.1145\/3770854.3783952","relation":{},"subject":[],"published":{"date-parts":[[2026,4,20]]},"assertion":[{"value":"2026-04-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}