{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T00:30:23Z","timestamp":1765499423095,"version":"3.48.0"},"publisher-location":"New York, NY, USA","reference-count":68,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["72171013, 72222022, 72242101, 62502404"],"award-info":[{"award-number":["72171013, 72222022, 72242101, 62502404"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["JKF-2025017226182"],"award-info":[{"award-number":["JKF-2025017226182"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Hong Kong Research Grants Council's Research Impact Fund","award":["R1015-23"],"award-info":[{"award-number":["R1015-23"]}]},{"name":"Collaborative Research Fund","award":["C1043-24GF"],"award-info":[{"award-number":["C1043-24GF"]}]},{"DOI":"10.13039\/501100012479","name":"General Research Fund","doi-asserted-by":"publisher","award":["11218325"],"award-info":[{"award-number":["11218325"]}],"id":[{"id":"10.13039\/501100012479","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Institute of Digital Medicine of City University of Hong Kong","award":["9229503"],"award-info":[{"award-number":["9229503"]}]},{"name":"Huawei","award":["Huawei Innovation Research Program"],"award-info":[{"award-number":["Huawei Innovation Research Program"]}]},{"DOI":"10.13039\/100015803","name":"Tencent","doi-asserted-by":"publisher","award":["CCF-Tencent Open Fund, Tencent Rhino-Bird Focused Research Program"],"award-info":[{"award-number":["CCF-Tencent Open Fund, Tencent Rhino-Bird Focused Research Program"]}],"id":[{"id":"10.13039\/100015803","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Alibaba","award":["CCF-Alimama Tech Kangaroo Fund No. 2024002"],"award-info":[{"award-number":["CCF-Alimama Tech Kangaroo Fund No. 2024002"]}]},{"DOI":"10.13039\/100018735","name":"Ant Group","doi-asserted-by":"publisher","award":["CCF-Ant Research Fund"],"award-info":[{"award-number":["CCF-Ant Research Fund"]}],"id":[{"id":"10.13039\/100018735","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Kuaishou"},{"name":"Didi","award":["CCF-Didi Gaia Scholars Research Fund"],"award-info":[{"award-number":["CCF-Didi Gaia Scholars Research Fund"]}]},{"name":"Bytedance"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,10]]},"DOI":"10.1145\/3746252.3761289","type":"proceedings-article","created":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T23:59:18Z","timestamp":1762559958000},"page":"2273-2283","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Contextual Attention Modulation: Towards Efficient Multi-Task Adaptation in Large Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-7712-8658","authenticated-orcid":false,"given":"Dayan","family":"Pan","sequence":"first","affiliation":[{"name":"SCSE, Beihang University, Beijing, China, ERC of ACAT, MOE, Beijing, China, and City University of Hong Kong, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7180-0977","authenticated-orcid":false,"given":"Zhaoyang","family":"Fu","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0651-1592","authenticated-orcid":false,"given":"Jingyuan","family":"Wang","sequence":"additional","affiliation":[{"name":"SCSE, Beihang University, Beijing, China, Key Lab of DIM, MIIT, Beijing, China, and SEM, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3478-964X","authenticated-orcid":false,"given":"Xiao","family":"Han","sequence":"additional","affiliation":[{"name":"Zhejiang University of Technology, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4776-5268","authenticated-orcid":false,"given":"Yue","family":"Zhu","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2926-4416","authenticated-orcid":false,"given":"Xiangyu","family":"Zhao","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,11,10]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_2_2_1","volume-title":"Jamie Ryan Kiros, and Geoffrey E Hinton","author":"Ba Jimmy Lei","year":"2016","unstructured":"Jimmy Lei Ba, Jamie Ryan Kiros, and Geoffrey E Hinton. 2016. Layer normalization. arXiv preprint arXiv:1607.06450 (2016)."},{"key":"e_1_3_2_2_3_1","unstructured":"Rishi Bommasani Drew A Hudson Ehsan Adeli Russ Altman Simran Arora Sydney von Arx Michael S Bernstein Jeannette Bohg Antoine Bosselut Emma Brunskill et al. 2021. On the opportunities and risks of foundation models. arXiv preprint arXiv:2108.07258 (2021)."},{"key":"e_1_3_2_2_4_1","volume-title":"Generative AI at work. The Quarterly Journal of Economics","author":"Brynjolfsson Erik","year":"2025","unstructured":"Erik Brynjolfsson, Danielle Li, and Lindsey Raymond. 2025. Generative AI at work. The Quarterly Journal of Economics (2025), qjae044."},{"key":"e_1_3_2_2_5_1","volume-title":"A survey on mixture of experts. arXiv preprint arXiv:2407.06204","author":"Cai Weilin","year":"2024","unstructured":"Weilin Cai, Juyong Jiang, Fan Wang, Jing Tang, Sunghun Kim, and Jiayi Huang. 2024. A survey on mixture of experts. arXiv preprint arXiv:2407.06204 (2024)."},{"key":"e_1_3_2_2_6_1","volume-title":"Code Alpaca: An Instruction-following LLaMA model for code generation. https:\/\/github.com\/sahil280114\/codealpaca.","author":"Chaudhary Sahil","year":"2023","unstructured":"Sahil Chaudhary. 2023. Code Alpaca: An Instruction-following LLaMA model for code generation. https:\/\/github.com\/sahil280114\/codealpaca."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i11.33252"},{"key":"e_1_3_2_2_8_1","volume-title":"Free Dolly: Introducing the World's First Truly Open Instruction-Tuned LLM. https:\/\/www.databricks.com\/blog\/2023\/04\/12\/dolly-first-open-commercially-viable-instruction-tuned-llm","author":"Conover Mike","year":"2023","unstructured":"Mike Conover, Matt Hayes, Ankit Mathur, Jianwei Xie, Jun Wan, Sam Shah, Ali Ghodsi, Patrick Wendell, Matei Zaharia, and Reynold Xin. 2023. Free Dolly: Introducing the World's First Truly Open Instruction-Tuned LLM. https:\/\/www.databricks.com\/blog\/2023\/04\/12\/dolly-first-open-commercially-viable-instruction-tuned-llm"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2021.3128667"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2017.12.012"},{"key":"e_1_3_2_2_11_1","first-page":"1","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus William","year":"2022","unstructured":"William Fedus, Barret Zoph, and Noam Shazeer. 2022. Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity. Journal of Machine Learning Research, Vol. 23, 120 (2022), 1-39.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_2_12_1","volume-title":"Sliding Window Attention Training for Efficient Large Language Models. arXiv preprint arXiv:2502.18845","author":"Fu Zichuan","year":"2025","unstructured":"Zichuan Fu, Wentao Song, Yejing Wang, Xian Wu, Yefeng Zheng, Yingying Zhang, Derong Xu, Xuetao Wei, Tong Xu, and Xiangyu Zhao. 2025a. Sliding Window Attention Training for Efficient Large Language Models. arXiv preprint arXiv:2502.18845 (2025)."},{"key":"e_1_3_2_2_13_1","volume-title":"Training-free LLM Merging for Multi-task Learning. arXiv preprint arXiv:2506.12379","author":"Fu Zichuan","year":"2025","unstructured":"Zichuan Fu, Xian Wu, Yejing Wang, Wanyu Wang, Shanshan Ye, Hongzhi Yin, Yi Chang, Yefeng Zheng, and Xiangyu Zhao. 2025b. Training-free LLM Merging for Multi-task Learning. arXiv preprint arXiv:2506.12379 (2025)."},{"key":"e_1_3_2_2_14_1","volume-title":"Higher layers need more lora experts. arXiv preprint arXiv:2402.08562","author":"Gao Chongyang","year":"2024","unstructured":"Chongyang Gao, Kezhen Chen, Jinmeng Rao, Baochen Sun, Ruibo Liu, Daiyi Peng, Yawen Zhang, Xiaoyuan Guo, Jie Yang, and VS Subrahmanian. 2024. Higher layers need more lora experts. arXiv preprint arXiv:2402.08562 (2024)."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.446"},{"key":"e_1_3_2_2_16_1","volume-title":"NLoRA: Nystr'' om-Initiated Low-Rank Adaptation for Large Language Models. arXiv preprint arXiv:2502.14482","author":"Guo Chenlu","year":"2025","unstructured":"Chenlu Guo, Yuan Wu, and Yi Chang. 2025. NLoRA: Nystr'' om-Initiated Low-Rank Adaptation for Large Language Models. arXiv preprint arXiv:2502.14482 (2025)."},{"key":"e_1_3_2_2_17_1","volume-title":"Large language model based multi-agents: A survey of progress and challenges. arXiv preprint arXiv:2402.01680","author":"Guo Taicheng","year":"2024","unstructured":"Taicheng Guo, Xiuying Chen, Yaqi Wang, Ruidi Chang, Shichao Pei, Nitesh V Chawla, Olaf Wiest, and Xiangliang Zhang. 2024. Large language model based multi-agents: A survey of progress and challenges. arXiv preprint arXiv:2402.01680 (2024)."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i11.33280"},{"key":"e_1_3_2_2_19_1","volume-title":"Parameter-efficient fine-tuning for large models: A comprehensive survey. arXiv preprint arXiv:2403.14608","author":"Han Zeyu","year":"2024","unstructured":"Zeyu Han, Chao Gao, Jinyang Liu, Jeff Zhang, and Sai Qian Zhang. 2024. Parameter-efficient fine-tuning for large models: A comprehensive survey. arXiv preprint arXiv:2403.14608 (2024)."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"e_1_3_2_2_21_1","volume-title":"Airphynet: Harnessing physics-guided neural networks for air quality prediction. arXiv preprint arXiv:2402.03784","author":"Hettige Kethmi Hirushini","year":"2024","unstructured":"Kethmi Hirushini Hettige, Jiahao Ji, Shili Xiang, Cheng Long, Gao Cong, and Jingyuan Wang. 2024. Airphynet: Harnessing physics-guided neural networks for air quality prediction. arXiv preprint arXiv:2402.03784 (2024)."},{"key":"e_1_3_2_2_22_1","volume-title":"International conference on machine learning. PMLR, 2790-2799","author":"Houlsby Neil","year":"2019","unstructured":"Neil Houlsby, Andrei Giurgiu, Stanislaw Jastrzebski, Bruna Morrone, Quentin De Laroussilhe, Andrea Gesmundo, Mona Attariyan, and Sylvain Gelly. 2019. Parameter-efficient transfer learning for NLP. In International conference on machine learning. PMLR, 2790-2799."},{"key":"e_1_3_2_2_23_1","volume-title":"Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685","author":"Hu Edward J","year":"2021","unstructured":"Edward J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2021. Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685 (2021)."},{"key":"e_1_3_2_2_24_1","volume-title":"Categorical reparameterization with gumbel-softmax. arXiv preprint arXiv:1611.01144","author":"Jang Eric","year":"2016","unstructured":"Eric Jang, Shixiang Gu, and Ben Poole. 2016. Categorical reparameterization with gumbel-softmax. arXiv preprint arXiv:1611.01144 (2016)."},{"key":"e_1_3_2_2_25_1","first-page":"9895","article-title":"Sparse is enough in scaling transformers","volume":"34","author":"Jaszczur Sebastian","year":"2021","unstructured":"Sebastian Jaszczur, Aakanksha Chowdhery, Afroz Mohiuddin, Lukasz Kaiser, Wojciech Gajewski, Henryk Michalewski, and Jonni Kanerva. 2021. Sparse is enough in scaling transformers. Advances in Neural Information Processing Systems, Vol. 34 (2021), 9895-9907.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_26_1","volume-title":"Multi-factor spatio-temporal prediction based on graph decomposition learning. arXiv preprint arXiv:2310.10374","author":"Ji Jiahao","year":"2023","unstructured":"Jiahao Ji, Jingyuan Wang, Yu Mou, and Cheng Long. 2023. Multi-factor spatio-temporal prediction based on graph decomposition learning. arXiv preprint arXiv:2310.10374 (2023)."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"crossref","unstructured":"Jiahao Ji Wentao Zhang Jingyuan Wang and Chao Huang. 2025. Seeing the unseen: Learning basis confounder representations for robust traffic prediction. (2025).","DOI":"10.1145\/3690624.3709201"},{"key":"e_1_3_2_2_28_1","volume-title":"Massive Values in Self-Attention Modules are the Key to Contextual Knowledge Understanding. arXiv preprint arXiv:2502.01563","author":"Jin Mingyu","year":"2025","unstructured":"Mingyu Jin, Kai Mei, Wujiang Xu, Mingjie Sun, Ruixiang Tang, Mengnan Du, Zirui Liu, and Yongfeng Zhang. 2025. Massive Values in Self-Attention Modules are the Key to Contextual Knowledge Understanding. arXiv preprint arXiv:2502.01563 (2025)."},{"key":"e_1_3_2_2_29_1","volume-title":"Gshard: Scaling giant models with conditional computation and automatic sharding. arXiv preprint arXiv:2006.16668","author":"Lepikhin Dmitry","year":"2020","unstructured":"Dmitry Lepikhin, HyoukJoong Lee, Yuanzhong Xu, Dehao Chen, Orhan Firat, Yanping Huang, Maxim Krikun, Noam Shazeer, and Zhifeng Chen. 2020. Gshard: Scaling giant models with conditional computation and automatic sharding. arXiv preprint arXiv:2006.16668 (2020)."},{"key":"e_1_3_2_2_30_1","volume-title":"The power of scale for parameter-efficient prompt tuning. arXiv preprint arXiv:2104.08691","author":"Lester Brian","year":"2021","unstructured":"Brian Lester, Rami Al-Rfou, and Noah Constant. 2021. The power of scale for parameter-efficient prompt tuning. arXiv preprint arXiv:2104.08691 (2021)."},{"key":"e_1_3_2_2_31_1","volume-title":"Jingyuan Wang, Jian-Yun Nie, and Ji-Rong Wen.","author":"Li Junyi","year":"2023","unstructured":"Junyi Li, Tianyi Tang, Wayne Xin Zhao, Jingyuan Wang, Jian-Yun Nie, and Ji-Rong Wen. 2023c. The web can be your oyster for improving large language models. arXiv preprint arXiv:2305.10998 (2023)."},{"key":"e_1_3_2_2_32_1","volume-title":"E4srec: An elegant effective efficient extensible solution of large language models for sequential recommendation. arXiv preprint arXiv:2312.02443","author":"Li Xinhang","year":"2023","unstructured":"Xinhang Li, Chong Chen, Xiangyu Zhao, Yong Zhang, and Chunxiao Xing. 2023a. E4srec: An elegant effective efficient extensible solution of large language models for sequential recommendation. arXiv preprint arXiv:2312.02443 (2023)."},{"key":"e_1_3_2_2_33_1","volume-title":"Prefix-tuning: Optimizing continuous prompts for generation. arXiv preprint arXiv:2101.00190","author":"Li Xiang Lisa","year":"2021","unstructured":"Xiang Lisa Li and Percy Liang. 2021. Prefix-tuning: Optimizing continuous prompts for generation. arXiv preprint arXiv:2101.00190 (2021)."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.7759\/cureus.40895"},{"key":"e_1_3_2_2_35_1","volume-title":"Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81.","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81."},{"key":"e_1_3_2_2_36_1","first-page":"18878","article-title":"Conflict-averse gradient descent for multi-task learning","volume":"34","author":"Liu Bo","year":"2021","unstructured":"Bo Liu, Xingchao Liu, Xiaojie Jin, Peter Stone, and Qiang Liu. 2021. Conflict-averse gradient descent for multi-task learning. Advances in Neural Information Processing Systems, Vol. 34 (2021), 18878-18890.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3711896.3737250"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657722"},{"key":"e_1_3_2_2_39_1","volume-title":"Kwang-Ting Cheng, and Min-Hung Chen.","author":"Liu Shih-Yang","year":"2024","unstructured":"Shih-Yang Liu, Chien-Yi Wang, Hongxu Yin, Pavlo Molchanov, Yu-Chiang Frank Wang, Kwang-Ting Cheng, and Min-Hung Chen. 2024a. Dora: Weight-decomposed low-rank adaptation. arXiv preprint arXiv:2402.09353 (2024)."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583467"},{"key":"e_1_3_2_2_41_1","volume-title":"Moelora: Contrastive learning guided mixture of experts on parameter-efficient fine-tuning for large language models. arXiv preprint arXiv:2402.12851","author":"Luo Tongxu","year":"2024","unstructured":"Tongxu Luo, Jiahe Lei, Fangyu Lei, Weihao Liu, Shizhu He, Jun Zhao, and Kang Liu. 2024. Moelora: Contrastive learning guided mixture of experts on parameter-efficient fine-tuning for large language models. arXiv preprint arXiv:2402.12851 (2024)."},{"key":"e_1_3_2_2_42_1","unstructured":"Reiichiro Nakano Jacob Hilton Suchir Balaji Jeff Wu Long Ouyang Christina Kim Christopher Hesse Shantanu Jain Vineet Kosaraju William Saunders Xu Jiang Karl Cobbe Tyna Eloundou Gretchen Krueger Kevin Button Matthew Knight Benjamin Chess and John Schulman. 2021. WebGPT: Browser-assisted question-answering with human feedback. In arXiv."},{"key":"e_1_3_2_2_43_1","volume-title":"Multi-task learning as a bargaining game. arXiv preprint arXiv:2202.01017","author":"Navon Aviv","year":"2022","unstructured":"Aviv Navon, Aviv Shamsian, Idan Achituve, Haggai Maron, Kenji Kawaguchi, Gal Chechik, and Ethan Fetaya. 2022. Multi-task learning as a bargaining game. arXiv preprint arXiv:2202.01017 (2022)."},{"key":"e_1_3_2_2_44_1","volume-title":"LISA: Layerwise Importance Sampling for Memory-Efficient Large Language Model Fine-Tuning. arXiv preprint arXiv:2403.17919","author":"Pan Rui","year":"2024","unstructured":"Rui Pan, Xiang Liu, Shizhe Diao, Renjie Pi, Jipeng Zhang, Chi Han, and Tong Zhang. 2024. LISA: Layerwise Importance Sampling for Memory-Efficient Large Language Model Fine-Tuning. arXiv preprint arXiv:2403.17919 (2024)."},{"key":"e_1_3_2_2_45_1","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311-318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311-318."},{"key":"e_1_3_2_2_46_1","volume-title":"Adapterfusion: Non-destructive task composition for transfer learning. arXiv preprint arXiv:2005.00247","author":"Pfeiffer Jonas","year":"2020","unstructured":"Jonas Pfeiffer, Aishwarya Kamath, Andreas R\u00fcckl\u00e9, Kyunghyun Cho, and Iryna Gurevych. 2020a. Adapterfusion: Non-destructive task composition for transfer learning. arXiv preprint arXiv:2005.00247 (2020)."},{"key":"e_1_3_2_2_47_1","volume-title":"Mad-x: An adapter-based framework for multi-task cross-lingual transfer. arXiv preprint arXiv:2005.00052","author":"Pfeiffer Jonas","year":"2020","unstructured":"Jonas Pfeiffer, Ivan Vuli\u0107, Iryna Gurevych, and Sebastian Ruder. 2020b. Mad-x: An adapter-based framework for multi-task cross-lingual transfer. arXiv preprint arXiv:2005.00052 (2020)."},{"key":"e_1_3_2_2_48_1","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter J Liu. 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. Journal of machine learning research, Vol. 21, 140 (2020), 1-67.","journal-title":"Journal of machine learning research"},{"key":"e_1_3_2_2_49_1","volume-title":"International conference on machine learning. PMLR","author":"Rajbhandari Samyam","year":"2022","unstructured":"Samyam Rajbhandari, Conglong Li, Zhewei Yao, Minjia Zhang, Reza Yazdani Aminabadi, Ammar Ahmad Awan, Jeff Rasley, and Yuxiong He. 2022. Deepspeed-moe: Advancing mixture-of-experts inference and training to power next-generation ai scale. In International conference on machine learning. PMLR, 18332-18346."},{"key":"e_1_3_2_2_50_1","volume-title":"A Stronger Mixture of Low-Rank Experts for Fine-Tuning Foundation Models. arXiv preprint arXiv:2502.15828","author":"Sun Mengyang","year":"2025","unstructured":"Mengyang Sun, Yihao Wang, Tao Feng, Dan Zhang, Yifan Zhu, and Jie Tang. 2025. A Stronger Mixture of Low-Rank Experts for Fine-Tuning Foundation Models. arXiv preprint arXiv:2502.15828 (2025)."},{"key":"e_1_3_2_2_51_1","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew M Dai Anja Hauth Katie Millican et al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"e_1_3_2_2_52_1","volume-title":"Attention is all you need. Advances in neural information processing systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599549"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3711896.3737257"},{"key":"e_1_3_2_2_55_1","volume-title":"MetaLoRA: Tensor-Enhanced Adaptive Low-Rank Fine-Tuning. In 2025 IEEE 41st International Conference on Data Engineering (ICDE). IEEE, 4680-4684","author":"Wang Maolin","year":"2025","unstructured":"Maolin Wang, Xiangyu Zhao, Ruocheng Guo, and Junhui Wang. 2025b. MetaLoRA: Tensor-Enhanced Adaptive Low-Rank Fine-Tuning. In 2025 IEEE 41st International Conference on Data Engineering (ICDE). IEEE, 4680-4684."},{"key":"e_1_3_2_2_56_1","volume-title":"2023 e. Large multimodal model compression via efficient pruning and distillation at AntGroup. arXiv preprint arXiv:2312.05795","author":"Wang Maolin","year":"2023","unstructured":"Maolin Wang, Yao Zhao, Jiajia Liu, Jingdong Chen, Chenyi Zhuang, Jinjie Gu, Ruocheng Guo, and Xiangyu Zhao. 2023 e. Large multimodal model compression via efficient pruning and distillation at AntGroup. arXiv preprint arXiv:2312.05795 (2023)."},{"key":"e_1_3_2_2_57_1","volume-title":"Jingyuan Wang, and Ji-Rong Wen.","author":"Wang Xiaolei","year":"2023","unstructured":"Xiaolei Wang, Xinyu Tang, Wayne Xin Zhao, Jingyuan Wang, and Ji-Rong Wen. 2023c. Rethinking the evaluation for conversational recommendation in the era of large language models. arXiv preprint arXiv:2305.13112 (2023)."},{"key":"e_1_3_2_2_58_1","volume-title":"Yi Wong, Ziru Liu, Xiangyu Zhao, Yichao Wang, Bo Chen, Huifeng Guo, and Ruiming Tang.","author":"Wang Yuhao","year":"2023","unstructured":"Yuhao Wang, Ha Tsz Lam, Yi Wong, Ziru Liu, Xiangyu Zhao, Yichao Wang, Bo Chen, Huifeng Guo, and Ruiming Tang. 2023a. Multi-task deep recommender systems: A survey. arXiv preprint arXiv:2302.03525 (2023)."},{"key":"e_1_3_2_2_59_1","volume-title":"Multilora: Democratizing lora for better multi-task learning. arXiv preprint arXiv:2311.11501","author":"Wang Yiming","year":"2023","unstructured":"Yiming Wang, Yu Lin, Xiaodong Zeng, and Guannan Zhang. 2023b. Multilora: Democratizing lora for better multi-task learning. arXiv preprint arXiv:2311.11501 (2023)."},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627673.3679743"},{"key":"e_1_3_2_2_61_1","volume-title":"Brian Lester, Nan Du, Andrew M Dai, and Quoc V Le.","author":"Wei Jason","year":"2021","unstructured":"Jason Wei, Maarten Bosma, Vincent Y Zhao, Kelvin Guu, Adams Wei Yu, Brian Lester, Nan Du, Andrew M Dai, and Quoc V Le. 2021. Finetuned language models are zero-shot learners. arXiv preprint arXiv:2109.01652 (2021)."},{"key":"e_1_3_2_2_62_1","first-page":"5824","article-title":"Gradient surgery for multi-task learning","volume":"33","author":"Yu Tianhe","year":"2020","unstructured":"Tianhe Yu, Saurabh Kumar, Abhishek Gupta, Sergey Levine, Karol Hausman, and Chelsea Finn. 2020. Gradient surgery for multi-task learning. Advances in Neural Information Processing Systems, Vol. 33 (2020), 5824-5836.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE65448.2025.00334"},{"key":"e_1_3_2_2_64_1","volume-title":"Bitfit: Simple parameter-efficient fine-tuning for transformer-based masked language-models. arXiv preprint arXiv:2106.10199","author":"Zaken Elad Ben","year":"2021","unstructured":"Elad Ben Zaken, Shauli Ravfogel, and Yoav Goldberg. 2021. Bitfit: Simple parameter-efficient fine-tuning for transformer-based masked language-models. arXiv preprint arXiv:2106.10199 (2021)."},{"key":"e_1_3_2_2_65_1","volume-title":"AdaLoRA: Adaptive budget allocation for parameter-efficient fine-tuning. arXiv preprint arXiv:2303.10512","author":"Zhang Qingru","year":"2023","unstructured":"Qingru Zhang, Minshuo Chen, Alexander Bukharin, Nikos Karampatziakis, Pengcheng He, Yu Cheng, Weizhu Chen, and Tuo Zhao. 2023a. AdaLoRA: Adaptive budget allocation for parameter-efficient fine-tuning. arXiv preprint arXiv:2303.10512 (2023)."},{"key":"e_1_3_2_2_66_1","unstructured":"Wentao Zhang Jingyuan Wang Yifan Yang et al. 2024. VecCity: A taxonomy-guided library for map entity representation learning. arXiv preprint arXiv:2411.00874 (2024)."},{"key":"e_1_3_2_2_67_1","volume-title":"Automatic Chain of Thought Prompting in Large Language Models. In The Eleventh International Conference on Learning Representations (ICLR","author":"Zhang Zhuosheng","year":"2023","unstructured":"Zhuosheng Zhang, Aston Zhang, Mu Li, and Alex Smola. 2023b. Automatic Chain of Thought Prompting in Large Language Models. In The Eleventh International Conference on Learning Representations (ICLR 2023)."},{"key":"e_1_3_2_2_68_1","unstructured":"Wayne Xin Zhao Kun Zhou Junyi Li Tianyi Tang Xiaolei Wang Yupeng Hou Yingqian Min Beichen Zhang Junjie Zhang Zican Dong et al. 2023. A survey of large language models. arXiv preprint arXiv:2303.18223 (2023)."}],"event":{"name":"CIKM '25: The 34th ACM International Conference on Information and Knowledge Management","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"],"location":"Seoul Republic of Korea","acronym":"CIKM '25"},"container-title":["Proceedings of the 34th ACM International Conference on Information and Knowledge Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746252.3761289","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T00:26:18Z","timestamp":1765499178000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746252.3761289"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,10]]},"references-count":68,"alternative-id":["10.1145\/3746252.3761289","10.1145\/3746252"],"URL":"https:\/\/doi.org\/10.1145\/3746252.3761289","relation":{},"subject":[],"published":{"date-parts":[[2025,11,10]]},"assertion":[{"value":"2025-11-10","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}