{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,7]],"date-time":"2026-08-07T11:06:58Z","timestamp":1786100818221,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":89,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,7,20]],"date-time":"2025-07-20T00:00:00Z","timestamp":1752969600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"Agriculture and Food Research Initiative (AFRI)","award":["No. 2020-67021-32799"],"award-info":[{"award-number":["No. 2020-67021-32799"]}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["No. IIS-2117902"],"award-info":[{"award-number":["No. IIS-2117902"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,7,20]]},"DOI":"10.1145\/3690624.3709196","type":"proceedings-article","created":{"date-parts":[[2025,4,4]],"date-time":"2025-04-04T18:42:22Z","timestamp":1743792142000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":9,"title":["ResMoE: Space-efficient Compression of Mixture of Experts LLMs via Residual Restoration"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-3878-6188","authenticated-orcid":false,"given":"Mengting","family":"Ai","sequence":"first","affiliation":[{"name":"UIUC, Champaign, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4450-2005","authenticated-orcid":false,"given":"Tianxin","family":"Wei","sequence":"additional","affiliation":[{"name":"UIUC, Champaign, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9462-9122","authenticated-orcid":false,"given":"Yifan","family":"Chen","sequence":"additional","affiliation":[{"name":"HKBU, Kowloon, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5534-3401","authenticated-orcid":false,"given":"Zhichen","family":"Zeng","sequence":"additional","affiliation":[{"name":"UIUC, Champaign, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1656-9165","authenticated-orcid":false,"given":"Ritchie","family":"Zhao","sequence":"additional","affiliation":[{"name":"NVIDIA, Seattle, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-4504-0717","authenticated-orcid":false,"given":"Girish","family":"Varatkar","sequence":"additional","affiliation":[{"name":"Apple, Cupertino, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8412-4320","authenticated-orcid":false,"given":"Bita Darvish","family":"Rouhani","sequence":"additional","affiliation":[{"name":"NVIDIA, Seattle, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1554-2761","authenticated-orcid":false,"given":"Xianfeng","family":"Tang","sequence":"additional","affiliation":[{"name":"Amazon, Palo Alto, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4405-3887","authenticated-orcid":false,"given":"Hanghang","family":"Tong","sequence":"additional","affiliation":[{"name":"UIUC, Champaign, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6429-6272","authenticated-orcid":false,"given":"Jingrui","family":"He","sequence":"additional","affiliation":[{"name":"UIUC, Champaign, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,7,20]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Mengting Ai Tianxin Wei Yifan Chen Zeming Guo and Jingrui He. 2025. MLP Fusion: Towards Efficient Fine-tuning of Dense and Mixture-of-Experts Language Models. arxiv: 2307.08941 [cs.LG] https:\/\/arxiv.org\/abs\/2307.08941"},{"key":"e_1_3_2_1_2_1","volume-title":"The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=CQsmMYmlP5T","author":"Ainsworth Samuel","year":"2023","unstructured":"Samuel Ainsworth, Jonathan Hayase, and Siddhartha Srinivasa. 2023. Git Re-Basin: Merging Models modulo Permutation Symmetries. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=CQsmMYmlP5T"},{"key":"e_1_3_2_1_3_1","unstructured":"Rohan Anil Andrew M Dai Orhan Firat Melvin Johnson Dmitry Lepikhin Alexandre Passos Siamak Shakeri Emanuel Taropa Paige Bailey Zhifeng Chen et al. 2023. Palm 2 technical report. arXiv preprint arXiv:2305.10403 (2023)."},{"key":"e_1_3_2_1_4_1","volume-title":"Efficient 8-Bit Quantization of Transformer Neural Machine Language Translation Model. arxiv","author":"Bhandare Aishwarya","year":"1906","unstructured":"Aishwarya Bhandare, Vamsi Sripathi, Deepthi Karkada, Vivek Menon, Sun Choi, Kushal Datta, and Vikram Saletore. 2019. Efficient 8-Bit Quantization of Transformer Neural Machine Language Translation Model. arxiv: 1906.00532 [cs.LG]"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6239"},{"key":"e_1_3_2_1_6_1","volume-title":"WAPITI: A Watermark for Finetuned Open-Source LLMs. arXiv preprint arXiv:2410.06467","author":"Chen Lingjie","year":"2024","unstructured":"Lingjie Chen, Ruizhong Qiu, Siyu Yuan, Zhining Liu, Tianxin Wei, Hyunsik Yoo, Zhichen Zeng, Deqing Yang, and Hanghang Tong. 2024. WAPITI: A Watermark for Finetuned Open-Source LLMs. arXiv preprint arXiv:2410.06467 (2024)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539452"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.5555\/3618408.3618615"},{"key":"e_1_3_2_1_9_1","volume-title":"Skyformer: Remodel self-attention with gaussian kernel and nystrbackslash'' om method. In Advances in Neural Information Processing Systems.","author":"Chen Yifan","year":"2021","unstructured":"Yifan Chen, Qi Zeng, Heng Ji, and Yun Yang. 2021. Skyformer: Remodel self-attention with gaussian kernel and nystrbackslash'' om method. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.5555\/2884435.2884456"},{"key":"e_1_3_2_1_11_1","volume-title":"International conference on machine learning. PMLR, 685--693","author":"Cuturi Marco","year":"2014","unstructured":"Marco Cuturi and Arnaud Doucet. 2014. Fast computation of Wasserstein barycenters. In International conference on machine learning. PMLR, 685--693."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"Damai Dai Chengqi Deng Chenggang Zhao R. X. Xu Huazuo Gao Deli Chen Jiashi Li Wangding Zeng Xingkai Yu Y. Wu Zhenda Xie Y. K. Li Panpan Huang Fuli Luo Chong Ruan Zhifang Sui and Wenfeng Liang. 2024. DeepSeekMoE: Towards Ultimate Expert Specialization in Mixture-of-Experts Language Models. arxiv: 2401.06066 [cs.CL]","DOI":"10.18653\/v1\/2024.acl-long.70"},{"key":"e_1_3_2_1_13_1","volume-title":"Weinberger (Eds.)","volume":"27","author":"Denton Emily L","year":"2014","unstructured":"Emily L Denton, Wojciech Zaremba, Joan Bruna, Yann LeCun, and Rob Fergus. 2014. Exploiting Linear Structure Within Convolutional Networks for Efficient Evaluation. In Advances in Neural Information Processing Systems, Z. Ghahramani, M. Welling, C. Cortes, N. Lawrence, and K.Q. Weinberger (Eds.), Vol. 27. Curran Associates, Inc. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2014\/file\/2afe4567e1bf64d32a5527244d104cea-Paper.pdf"},{"key":"e_1_3_2_1_14_1","unstructured":"Tim Dettmers Mike Lewis Younes Belkada and Luke Zettlemoyer. 2022. LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale. arxiv: 2208.07339 [cs.LG]"},{"key":"e_1_3_2_1_15_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arxiv","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arxiv: 1810.04805 [cs.CL]"},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings of the Third International Workshop on Paraphrasing (IWP2005)","author":"William","unstructured":"William B. Dolan and Chris Brockett. 2005. Automatically Constructing a Corpus of Sentential Paraphrases. In Proceedings of the Third International Workshop on Paraphrasing (IWP2005). https:\/\/aclanthology.org\/I05--5002"},{"key":"e_1_3_2_1_17_1","volume-title":"LORAMOE: REVOLUTIONIZING MIXTURE OF EX-PERTS FOR MAINTAINING WORLD KNOWLEDGE IN LANGUAGE MODEL ALIGNMENT. arXiv preprint arXiv:2312.09979","author":"Dou Shihan","year":"2023","unstructured":"Shihan Dou, Enyu Zhou, Yan Liu, Songyang Gao, Jun Zhao, Wei Shen, Yuhao Zhou, Zhiheng Xi, Xiao Wang, Xiaoran Fan, et al. 2023. LORAMOE: REVOLUTIONIZING MIXTURE OF EX-PERTS FOR MAINTAINING WORLD KNOWLEDGE IN LANGUAGE MODEL ALIGNMENT. arXiv preprint arXiv:2312.09979 (2023)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.5555\/3586589.3586709"},{"key":"e_1_3_2_1_20_1","volume-title":"The lottery ticket hypothesis: Finding sparse, trainable neural networks. arXiv preprint arXiv:1803.03635","author":"Frankle Jonathan","year":"2018","unstructured":"Jonathan Frankle and Michael Carbin. 2018. The lottery ticket hypothesis: Finding sparse, trainable neural networks. arXiv preprint arXiv:1803.03635 (2018)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525243"},{"key":"e_1_3_2_1_22_1","volume-title":"GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers. arxiv: 2210.17323 [cs.LG]","author":"Frantar Elias","year":"2023","unstructured":"Elias Frantar, Saleh Ashkboos, Torsten Hoefler, and Dan Alistarh. 2023. GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers. arxiv: 2210.17323 [cs.LG]"},{"key":"e_1_3_2_1_23_1","volume-title":"Zhong-Yi Lu, and Ji-Rong Wen.","author":"Gao Ze-Feng","year":"2022","unstructured":"Ze-Feng Gao, Peiyu Liu, Wayne Xin Zhao, Zhong-Yi Lu, and Ji-Rong Wen. 2022. Parameter-Efficient Mixture-of-Experts Architecture for Pre-trained Language Models. In Proceedings of the 29th International Conference on Computational Linguistics, Nicoletta Calzolari, Chu-Ren Huang, Hansaem Kim, James Pustejovsky, Leo Wanner, Key-Sun Choi, Pum-Mo Ryu, Hsin-Hsi Chen, Lucia Donatelli, Heng Ji, Sadao Kurohashi, Patrizia Paggio, Nianwen Xue, Seokhwan Kim, Younggyun Hahm, Zhong He, Tony Kyungil Lee, Enrico Santus, Francis Bond, and Seung-Hoon Na (Eds.). International Committee on Computational Linguistics, Gyeongju, Republic of Korea, 3263--3273. https:\/\/aclanthology.org\/2022.coling-1.288"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01453-z"},{"key":"e_1_3_2_1_25_1","volume-title":"Garnett (Eds.)","volume":"28","author":"Han Song","year":"2015","unstructured":"Song Han, Jeff Pool, John Tran, and William Dally. 2015. Learning both Weights and Connections for Efficient Neural Network. In Advances in Neural Information Processing Systems, C. Cortes, N. Lawrence, D. Lee, M. Sugiyama, and R. Garnett (Eds.), Vol. 28. Curran Associates, Inc. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2015\/file\/ae0eb3eed39d2bcef4622b2499a05fe6-Paper.pdf"},{"key":"e_1_3_2_1_26_1","volume-title":"Preserving Pre-trained Features Helps Calibrate Fine-tuned Language Models. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=NI7StoWHJPT","author":"He Guande","year":"2023","unstructured":"Guande He, Jianfei Chen, and Jun Zhu. 2023a. Preserving Pre-trained Features Helps Calibrate Fine-tuned Language Models. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=NI7StoWHJPT"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539315"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.907"},{"key":"e_1_3_2_1_29_1","volume-title":"LLM-Forest for Health Tabular Data Imputation. arXiv preprint arXiv:2410.21520","author":"He Xinrui","year":"2024","unstructured":"Xinrui He, Yikun Ban, Jiaru Zou, Tianxin Wei, Curtiss B Cook, and Jingrui He. 2024. LLM-Forest for Health Tabular Data Imputation. arXiv preprint arXiv:2410.21520 (2024)."},{"key":"e_1_3_2_1_30_1","unstructured":"Albert Q. Jiang Alexandre Sablayrolles Arthur Mensch Chris Bamford Devendra Singh Chaplot Diego de las Casas Florian Bressand Gianna Lengyel Guillaume Lample Lucile Saulnier L\u00e9lio Renard Lavaud Marie-Anne Lachaux Pierre Stock Teven Le Scao Thibaut Lavril Thomas Wang Timoth\u00e9e Lacroix and William El Sayed. 2023. Mistral 7B. arxiv: 2310.06825 [cs.CL]"},{"key":"e_1_3_2_1_31_1","unstructured":"Albert Q. Jiang Alexandre Sablayrolles Antoine Roux Arthur Mensch Blanche Savary Chris Bamford Devendra Singh Chaplot Diego de las Casas Emma Bou Hanna Florian Bressand Gianna Lengyel Guillaume Bour Guillaume Lample L\u00e9lio Renard Lavaud Lucile Saulnier Marie-Anne Lachaux Pierre Stock Sandeep Subramanian Sophia Yang Szymon Antoniak Teven Le Scao Th\u00e9ophile Gervet Thibaut Lavril Thomas Wang Timoth\u00e9e Lacroix and William El Sayed. 2024. Mixtral of Experts. arxiv: 2401.04088 [cs.LG]"},{"key":"e_1_3_2_1_32_1","unstructured":"Bowen Jin Hansi Zeng Guoyin Wang Xiusi Chen Tianxin Wei Ruirui Li Zhengyang Wang Zheng Li Yang Li Hanqing Lu et al. 2023. Language models as semantic indexers. arXiv preprint arXiv:2310.07815 (2023)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3340531.3412005"},{"key":"e_1_3_2_1_34_1","unstructured":"Rui Kong Yuanchun Li Qingtian Feng Weijun Wang Linghe Kong and Yunxin Liu. 2023. SwapMoE: Efficient Memory-Constrained Serving of Large Sparse MoE Models via Dynamic Expert Pruning and Swapping. arxiv: 2308.15030 [cs.AI]"},{"key":"e_1_3_2_1_35_1","volume-title":"SNIP: SINGLE-SHOT NETWORK PRUNING BASED ON CONNECTION SENSITIVITY. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=B1VZqjAcYX","author":"Lee Namhoon","year":"2019","unstructured":"Namhoon Lee, Thalaiyasingam Ajanthan, and Philip Torr. 2019. SNIP: SINGLE-SHOT NETWORK PRUNING BASED ON CONNECTION SENSITIVITY. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=B1VZqjAcYX"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599284"},{"key":"e_1_3_2_1_37_1","volume-title":"Merge","author":"Li Pingzhi","year":"2023","unstructured":"Pingzhi Li, Zhenyu Zhang, Prateek Yadav, Yi-Lin Sung, Yu Cheng, Mohit Bansal, and Tianlong Chen. 2023c. Merge, Then Compress: Demystify Efficient SMoE with Hints from Its Routing Policy. arXiv preprint arXiv:2310.01334 (2023)."},{"key":"e_1_3_2_1_38_1","unstructured":"Yixiao Li Yifan Yu Qingru Zhang Chen Liang Pengcheng He Weizhu Chen and Tuo Zhao. 2023a. LoSparse: Structured Compression of Large Language Models based on Low-Rank and Sparse Approximation. arxiv: 2306.11222 [cs.LG]"},{"key":"e_1_3_2_1_39_1","unstructured":"Zihao Li Dongqi Fu Mengting Ai and Jingrui He. 2024. APEX^2: Adaptive and Extreme Summarization for Personalized Knowledge Graphs. arxiv: 2412.17336 [cs.LG] https:\/\/arxiv.org\/abs\/2412.17336"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-short.22"},{"key":"e_1_3_2_1_41_1","unstructured":"Jiacheng Lin Kun Qian Haoyu Han Nurendra Choudhary Tianxin Wei Zhongruo Wang Sahika Genc Edward W Huang Sheng Wang Karthik Subbian et al. 2024. Unleashing the Power of LLMs as Multi-Modal Encoders for Text and Graph-Structured Data. arXiv preprint arXiv:2410.11235 (2024)."},{"key":"e_1_3_2_1_42_1","volume-title":"Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"13869","author":"Liu Chang","year":"2022","unstructured":"Chang Liu, Chenfei Lou, Runzhong Wang, Alan Yuhan Xi, Li Shen, and Junchi Yan. 2022. Deep Neural Network Fusion via Graph Matching with Applications to Model Ensemble and Federated Learning. In Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 162), Kamalika Chaudhuri, Stefanie Jegelka, Le Song, Csaba Szepesvari, Gang Niu, and Sivan Sabato (Eds.). PMLR, 13857--13869. https:\/\/proceedings.mlr.press\/v162\/liu22k.html"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_44_1","volume-title":"Rethinking the value of network pruning. arXiv preprint arXiv:1810.05270","author":"Liu Zhuang","year":"2018","unstructured":"Zhuang Liu, Mingjie Sun, Tinghui Zhou, Gao Huang, and Trevor Darrell. 2018. Rethinking the value of network pruning. arXiv preprint arXiv:1810.05270 (2018)."},{"key":"e_1_3_2_1_45_1","unstructured":"Xudong Lu Qi Liu Yuhui Xu Aojun Zhou Siyuan Huang Bo Zhang Junchi Yan and Hongsheng Li. 2024. Not All Experts are Equal: Efficient Expert Pruning and Skipping for Mixture-of-Experts Large Language Models. arxiv: 2402.14800 [cs.CL] https:\/\/arxiv.org\/abs\/2402.14800"},{"key":"e_1_3_2_1_46_1","volume-title":"Pointer sentinel mixture models. arXiv preprint arXiv:1609.07843","author":"Merity Stephen","year":"2016","unstructured":"Stephen Merity, Caiming Xiong, James Bradbury, and Richard Socher. 2016. Pointer sentinel mixture models. arXiv preprint arXiv:1609.07843 (2016)."},{"key":"e_1_3_2_1_47_1","volume-title":"Dokania","author":"Mukhoti Jishnu","year":"2023","unstructured":"Jishnu Mukhoti, Yarin Gal, Philip H. S. Torr, and Puneet K. Dokania. 2023. Fine-tuning can cripple your foundation model; preserving features may be the solution. arxiv: 2308.13320 [cs.LG]"},{"key":"e_1_3_2_1_48_1","unstructured":"Alexandre Muzio Alex Sun and Churan He. 2024. SEER-MoE: Sparse Expert Efficiency through Regularization for Mixture-of-Experts. arxiv: 2404.05089 [cs.CL] https:\/\/arxiv.org\/abs\/2404.05089"},{"key":"e_1_3_2_1_49_1","unstructured":"OpenAI. 2022. Techniques for training large neural networks. https:\/\/openai.com\/research\/techniques-for-training-large-neural-networks"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-1144"},{"key":"e_1_3_2_1_51_1","volume-title":"Computational Optimal Transport. arxiv","author":"Peyre Gabriel","year":"1803","unstructured":"Gabriel Peyre and Marco Cuturi. 2020. Computational Optimal Transport. arxiv: 1803.00567 [stat.ML]"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599371"},{"key":"e_1_3_2_1_53_1","volume-title":"Gradient Compressed Sensing: A Query-Efficient Gradient Estimator for High-Dimensional Zeroth-Order Optimization. In Forty-first International Conference on Machine Learning.","author":"Qiu Ruizhong","unstructured":"Ruizhong Qiu and Hanghang Tong. [n.,d.]. Gradient Compressed Sensing: A Query-Efficient Gradient Estimator for High-Dimensional Zeroth-Order Optimization. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_1_54_1","volume-title":"Turing completeness of prompting. arXiv preprint arXiv:2411.01992","author":"Qiu Ruizhong","year":"2024","unstructured":"Ruizhong Qiu, Zhe Xu, Wenxuan Bao, and Hanghang Tong. 2024a. Ask, and it shall be given: Turing completeness of prompting. arXiv preprint arXiv:2411.01992 (2024)."},{"key":"e_1_3_2_1_55_1","volume-title":"Hanghang Tong, James Ezick, and Christopher Lott.","author":"Qiu Ruizhong","year":"2024","unstructured":"Ruizhong Qiu, Weiliang Will Zeng, Hanghang Tong, James Ezick, and Christopher Lott. 2024b. How Efficient is LLM-Generated Code? A Rigorous & High-Standard Benchmark. arXiv preprint arXiv:2406.06647 (2024)."},{"key":"e_1_3_2_1_56_1","volume-title":"Liu","author":"Raffel Colin","year":"2023","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter J. Liu. 2023. Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer. arxiv: 1910.10683 [cs.LG] https:\/\/arxiv.org\/abs\/1910.10683"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474381"},{"key":"e_1_3_2_1_58_1","unstructured":"Pratyusha Sharma Jordan T. Ash and Dipendra Misra. 2023. The Truth is in There: Improving Reasoning in Language Models with Layer-Selective Rank Reduction. arxiv: 2312.13558 [cs.LG]"},{"key":"e_1_3_2_1_59_1","unstructured":"Noam Shazeer Azalia Mirhoseini Krzysztof Maziarz Andy Davis Quoc Le Geoffrey Hinton and Jeff Dean. 2017. Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer. arxiv: 1701.06538 [cs.LG]"},{"key":"e_1_3_2_1_60_1","first-page":"22045","article-title":"Model fusion via optimal transport","volume":"33","author":"Singh Sidak Pal","year":"2020","unstructured":"Sidak Pal Singh and Martin Jaggi. 2020. Model fusion via optimal transport. Advances in Neural Information Processing Systems, Vol. 33 (2020), 22045--22055.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D13-1170"},{"key":"e_1_3_2_1_62_1","unstructured":"George Stoica Daniel Bolya Jakob Bjorner Pratik Ramesh Taylor Hearn and Judy Hoffman. 2024. ZipIt! Merging Models from Different Tasks without Training. arxiv: 2305.03053 [cs.CV] https:\/\/arxiv.org\/abs\/2305.03053"},{"key":"e_1_3_2_1_63_1","volume-title":"The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=PxoFut3dWW","author":"Sun Mingjie","year":"2024","unstructured":"Mingjie Sun, Zhuang Liu, Anna Bair, and J Zico Kolter. 2024. A Simple and Effective Pruning Approach for Large Language Models. In The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=PxoFut3dWW"},{"key":"e_1_3_2_1_64_1","volume-title":"Lin (Eds.)","volume":"33","author":"Tanaka Hidenori","year":"2020","unstructured":"Hidenori Tanaka, Daniel Kunin, Daniel L Yamins, and Surya Ganguli. 2020. Pruning neural networks without any data by iteratively conserving synaptic flow. In Advances in Neural Information Processing Systems, H. Larochelle, M. Ranzato, R. Hadsell, M.F. Balcan, and H. Lin (Eds.), Vol. 33. Curran Associates, Inc., 6377--6389. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2020\/file\/46a4378f835dc8040c8057beb6a2da52-Paper.pdf"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.331"},{"key":"e_1_3_2_1_66_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale Dan Bikel Lukas Blecher Cristian Canton Ferrer Moya Chen Guillem Cucurull David Esiobu Jude Fernandes Jeremy Fu Wenyin Fu Brian Fuller Cynthia Gao Vedanuj Goswami Naman Goyal Anthony Hartshorn Saghar Hosseini Rui Hou Hakan Inan Marcin Kardas Viktor Kerkez Madian Khabsa Isabel Kloumann Artem Korenev Punit Singh Koura Marie-Anne Lachaux Thibaut Lavril Jenya Lee Diana Liskovich Yinghai Lu Yuning Mao Xavier Martinet Todor Mihaylov Pushkar Mishra Igor Molybog Yixin Nie Andrew Poulton Jeremy Reizenstein Rashi Rungta Kalyan Saladi Alan Schelten Ruan Silva Eric Michael Smith Ranjan Subramanian Xiaoqing Ellen Tan Binh Tang Ross Taylor Adina Williams Jian Xiang Kuan Puxin Xu Zheng Yan Iliyan Zarov Yuchen Zhang Angela Fan Melanie Kambadur Sharan Narang Aurelien Rodriguez Robert Stojnic Sergey Edunov and Thomas Scialom. 2023. Llama 2: Open Foundation and Fine-Tuned Chat Models. arxiv: 2307.09288 [cs.CL]"},{"key":"e_1_3_2_1_67_1","volume-title":"\u0141 ukasz Kaiser, and Illia Polosukhin","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141 ukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Advances in Neural Information Processing Systems, I. Guyon, U. Von Luxburg, S. Bengio, H. Wallach, R. Fergus, S. Vishwanathan, and R. Garnett (Eds.), Vol. 30. Curran Associates, Inc. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2017\/file\/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf"},{"key":"e_1_3_2_1_68_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=RftryyYyjiG","author":"Wang Benyou","year":"2022","unstructured":"Benyou Wang, Yuxin Ren, Lifeng Shang, Xin Jiang, and Qun Liu. 2022. Exploring extreme parameter compression for pre-trained language models. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=RftryyYyjiG"},{"key":"e_1_3_2_1_69_1","volume-title":"Picking winning tickets before training by preserving gradient flow. arXiv preprint arXiv:2002.07376","author":"Wang Chaoqi","year":"2020","unstructured":"Chaoqi Wang, Guodong Zhang, and Roger Grosse. 2020. Picking winning tickets before training by preserving gradient flow. arXiv preprint arXiv:2002.07376 (2020)."},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00061"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00290"},{"key":"e_1_3_2_1_72_1","unstructured":"Tianxin Wei Yifan Chen Xinrui He and Jingrui He. 2024a. Connecting Domains and Contrasting Samples: A Ladder for Domain Generalization. (2024)."},{"key":"e_1_3_2_1_73_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"36838","author":"Wei Tianxin","year":"2023","unstructured":"Tianxin Wei, Zeming Guo, Yifan Chen, and Jingrui He. 2023. NTK-approximating MLP Fusion for Efficient Language Model Fine-tuning. In Proceedings of the 40th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 202), Andreas Krause, Emma Brunskill, Kyunghyun Cho, Barbara Engelhardt, Sivan Sabato, and Jonathan Scarlett (Eds.). PMLR, 36821--36838. https:\/\/proceedings.mlr.press\/v202\/wei23b.html"},{"key":"e_1_3_2_1_74_1","unstructured":"Tianxin Wei Bowen Jin Ruirui Li Hansi Zeng Zhengyang Wang Jianhui Sun Qingyu Yin Hanqing Lu Suhang Wang Jingrui He et al. 2024b. Towards unified multi-modal personalization: Large vision-language models for generative recommendation and beyond. arXiv preprint arXiv:2403.10667 (2024)."},{"key":"e_1_3_2_1_75_1","unstructured":"Tianxin Wei Ruizhong Qiu Yifan Chen Yunzhe Qi Jiacheng Lin Wenju Xu Sreyashi Nag Ruirui Li Hanqing Lu Zhengyang Wang et al. 2024c. Robust Watermarking for Diffusion Models: A Unified Multi-Dimensional Recipe. (2024)."},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-1101"},{"key":"e_1_3_2_1_77_1","volume-title":"Language Models are Graph Learners. arXiv preprint arXiv:2410.02296","author":"Xu Zhe","year":"2024","unstructured":"Zhe Xu, Kaveh Hassani, Si Zhang, Hanqing Zeng, Michihiro Yasunaga, Limei Wang, Dongqi Fu, Ning Yao, Bo Long, and Hanghang Tong. 2024. Language Models are Graph Learners. arXiv preprint arXiv:2410.02296 (2024)."},{"key":"e_1_3_2_1_78_1","unstructured":"Fuzhao Xue Xiaoxin He Xiaozhe Ren Yuxuan Lou and Yang You. 2022. One Student Knows All Experts Know: From Sparse to Dense. arxiv: 2201.10890 [cs.LG]"},{"key":"e_1_3_2_1_79_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Yan Yuchen","year":"2024","unstructured":"Yuchen Yan, Baoyu Jing, Lihui Liu, Ruijie Wang, Jinning Li, Tarek Abdelzaher, and Hanghang Tong. 2024. Reconciling competing sampling strategies of network embedding. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_80_1","doi-asserted-by":"publisher","DOI":"10.14778\/3529337.3529343"},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645477"},{"key":"e_1_3_2_1_82_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i15.29605"},{"key":"e_1_3_2_1_83_1","unstructured":"Zhichen Zeng Xiaolong Liu Mengyue Hang Xiaoyi Liu Qinghai Zhou Chaofei Yang Yiqun Liu Yichen Ruan Laming Chen Yuxin Chen et al. 2024b. InterFormer: Towards Effective Heterogeneous Interaction Learning for Click-Through Rate Prediction. arXiv preprint arXiv:2411.09852 (2024)."},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583357"},{"key":"e_1_3_2_1_85_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3615499"},{"key":"e_1_3_2_1_86_1","volume-title":"Knowledge overshadowing causes amalgamated hallucination in large language models. arXiv preprint arXiv:2407.08039","author":"Zhang Yuji","year":"2024","unstructured":"Yuji Zhang, Sha Li, Jiateng Liu, Pengfei Yu, Yi R Fung, Jing Li, Manling Li, and Heng Ji. 2024a. Knowledge overshadowing causes amalgamated hallucination in large language models. arXiv preprint arXiv:2407.08039 (2024)."},{"key":"e_1_3_2_1_87_1","doi-asserted-by":"publisher","DOI":"10.1128\/spectrum.02039-23"},{"key":"e_1_3_2_1_88_1","volume-title":"STEM-POM: Evaluating Language Models Math-Symbol Reasoning in Document Parsing. arXiv preprint arXiv:2411.00387","author":"Zou Jiaru","year":"2024","unstructured":"Jiaru Zou, Qing Wang, Pratyush Thakur, and Nickvash Kani. 2024a. STEM-POM: Evaluating Language Models Math-Symbol Reasoning in Document Parsing. arXiv preprint arXiv:2411.00387 (2024)."},{"key":"e_1_3_2_1_89_1","volume-title":"Promptintern: Saving inference costs by internalizing recurrent prompt during large language model fine-tuning. arXiv preprint arXiv:2407.02211","author":"Zou Jiaru","year":"2024","unstructured":"Jiaru Zou, Mengyu Zhou, Tao Li, Shi Han, and Dongmei Zhang. 2024b. Promptintern: Saving inference costs by internalizing recurrent prompt during large language model fine-tuning. arXiv preprint arXiv:2407.02211 (2024)."}],"event":{"name":"KDD '25: The 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Toronto ON Canada","acronym":"KDD '25","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.1"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3690624.3709196","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3690624.3709196","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,16]],"date-time":"2025-08-16T15:33:29Z","timestamp":1755358409000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3690624.3709196"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,20]]},"references-count":89,"alternative-id":["10.1145\/3690624.3709196","10.1145\/3690624"],"URL":"https:\/\/doi.org\/10.1145\/3690624.3709196","relation":{},"subject":[],"published":{"date-parts":[[2025,7,20]]},"assertion":[{"value":"2025-07-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}