{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,28]],"date-time":"2026-08-28T16:52:20Z","timestamp":1787935940308,"version":"build-2784847793"},"reference-count":182,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Knowl. Data Eng."],"published-print":{"date-parts":[[2025]]},"DOI":"10.1109\/tkde.2025.3554028","type":"journal-article","created":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T03:46:33Z","timestamp":1742960793000},"page":"1-20","source":"Crossref","is-referenced-by-count":143,"title":["A Survey on Mixture of Experts in Large Language Models"],"prefix":"10.1109","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6369-6389","authenticated-orcid":false,"given":"Weilin","family":"Cai","sequence":"first","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Juyong","family":"Jiang","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fan","family":"Wang","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0785-707X","authenticated-orcid":false,"given":"Jing","family":"Tang","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sunghun","family":"Kim","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4011-6668","authenticated-orcid":false,"given":"Jiayi","family":"Huang","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref2","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Brown","year":"2020"},{"issue":"240","key":"ref3","first-page":"1","article-title":"PaLM: Scaling language modeling with pathways","volume":"24","author":"Chowdhery","year":"2023","journal-title":"J. Mach. Learn. Res."},{"key":"ref4","article-title":"GPT-4 technical report","author":"Achiam","year":"2023"},{"key":"ref5","article-title":"A survey on large language models for code generation","author":"Jiang","year":"2024"},{"key":"ref6","first-page":"8583","article-title":"Scaling vision with sparse mixture of experts","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Riquelme","year":"2021"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref8","first-page":"13","article-title":"ViLBERT: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Lu","year":"2019"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01653-1"},{"key":"ref10","article-title":"MiniGPT-4: Enhancing vision-language understanding with advanced large language models","author":"Zhu","year":"2023"},{"key":"ref11","article-title":"Scaling laws for neural language models","author":"Kaplan","year":"2020"},{"key":"ref12","article-title":"Emergent abilities of large language models","author":"Wei","year":"2022"},{"key":"ref13","article-title":"HyperCLOVA X technical report","author":"Yoo","year":"2024"},{"key":"ref14","article-title":"Training compute-optimal large language models","author":"Hoffmann","year":"2022"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1991.3.1.79"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1994.6.2.181"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/1120.003.0086"},{"key":"ref18","first-page":"881","article-title":"Infinite mixtures of Gaussian process experts","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Rasmussen","year":"2001"},{"issue":"8","key":"ref19","first-page":"1829","article-title":"Nonlinear models using Dirichlet process mixtures","volume":"10","author":"Shahbaba","year":"2009","journal-title":"J. Mach. Learn. Res."},{"key":"ref20","article-title":"Learning factored representations in a deep mixture of experts","author":"Eigen","year":"2013"},{"key":"ref21","first-page":"1927","article-title":"Generative image modeling using spatial LSTMs","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Theis","year":"2015"},{"key":"ref22","first-page":"1481","article-title":"Distributed Gaussian processes","volume-title":"Proc. 32nd Int. Conf. Mach. Learn.","author":"Deisenroth","year":"2015"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.753"},{"key":"ref24","article-title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer","author":"Shazeer","year":"2017"},{"key":"ref25","article-title":"GShard: Scaling giant models with conditional computation and automatic sharding","author":"Lepikhin","year":"2020"},{"key":"ref26","article-title":"Mixtral of experts","author":"Jiang","year":"2024"},{"key":"ref27","article-title":"Grok-1","year":"2024"},{"key":"ref28","article-title":"Introducing DBRX: A new State-of-the-Art open LLM","year":"2024"},{"key":"ref29","article-title":"Snowflake arctic: The best LLM for enterprise AI \u2014 Efficiently intelligent, truly open","year":"2024"},{"key":"ref30","article-title":"Deepseek-v2: A strong, economical, and efficient mixture-of-experts language model","author":"Liu","year":"2024"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2012.2200299"},{"key":"ref32","article-title":"A review of sparse expert models in deep learning","author":"Fedus","year":"2022"},{"key":"ref33","first-page":"5547","article-title":"GLaM: Efficient scaling of language models with mixture-of-experts","volume-title":"Proc. 39th Int. Conf. Mach. Learn.","author":"Du","year":"2022"},{"issue":"120","key":"ref34","first-page":"1","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus","year":"2022","journal-title":"J. Mach. Learn. Res."},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ipdpsw55747.2022.00171"},{"key":"ref36","article-title":"OpenMoE: An early effort on open mixture-of-experts language models","author":"Xue","year":"2024"},{"key":"ref37","article-title":"From sparse to soft mixtures of experts","volume-title":"Proc. 12th Int. Conf. Learn. Representations","author":"Puigcerver","year":"2023"},{"key":"ref38","article-title":"Soft merging of experts with adaptive routing","author":"Muqeeth","year":"2023"},{"key":"ref39","article-title":"Lory: Fully differentiable mixture-of-experts for autoregressive language model pre-training","author":"Zhong","year":"2024"},{"key":"ref40","article-title":"Pushing mixture of experts to the limit: Extremely parameter efficient MoE for instruction tuning","author":"Zadouri","year":"2023"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.01347"},{"key":"ref42","first-page":"5744","article-title":"AdaMix: Mixture-of-adaptations for parameter-efficient model tuning","volume-title":"Proc. Conf. Empirical Methods Natural Lang. Process.","author":"Wang","year":"2022"},{"key":"ref43","article-title":"LoRAMoE: Revolutionizing mixture of experts for maintaining world knowledge in language model alignment","author":"Dou","year":"2023"},{"key":"ref44","article-title":"Mixture of cluster-conditional LoRA experts for vision-language instruction tuning","author":"Gou","year":"2023"},{"key":"ref45","article-title":"MoELoRA: Contrastive learning guided mixture of experts on parameter-efficient fine-tuning for large language models","author":"Luo","year":"2024"},{"key":"ref46","article-title":"Mixture of LoRA experts","volume-title":"Proc. 12th Int. Conf. Learn. Representations","author":"Wu","year":"2024"},{"key":"ref47","article-title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","volume-title":"Proc. 11th Int. Conf. Learn. Representations","author":"Komatsuzaki","year":"2022"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.71"},{"key":"ref49","article-title":"LLaMA-MoE: Building mixture-of-experts from LLaMA with continual pre-training","year":"2023"},{"key":"ref50","article-title":"One student knows all experts know: From sparse to dense","author":"Xue","year":"2022"},{"key":"ref51","article-title":"Task-specific expert pruning for sparse mixture-of-experts","author":"Chen","year":"2022"},{"key":"ref52","article-title":"Branch-train-MiX: Mixing expert LLMs into a mixture-of-experts LLM","author":"Sukhbaatar","year":"2024"},{"key":"ref53","first-page":"5383","article-title":"Lifelong language pretraining with distribution-specialized experts","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","author":"Chen","year":"2023"},{"key":"ref54","article-title":"Mixture of tokens: Efficient LLMs through cross-example aggregation","author":"Antoniak","year":"2023"},{"key":"ref55","article-title":"Mixture-of-depths: Dynamically allocating compute in transformer-based language models","author":"Raposo","year":"2024"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i8.20858"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.12"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.884"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220007"},{"key":"ref60","article-title":"EvoMoE: An evolutional mixture-of-experts training framework via dense-to-sparse gate","author":"Nie","year":"2021"},{"key":"ref61","article-title":"MoLE: Mixture of LoRA experts","volume-title":"Proc. 12th Int. Conf. Learn. Representations","author":"Wu","year":"2023"},{"key":"ref62","article-title":"Dense training, sparse inference: Rethinking training of mixture-of-experts language models","author":"Pan","year":"2024"},{"key":"ref63","first-page":"4057","article-title":"Unified scaling laws for routed language models","volume-title":"Proc. 39th Int. Conf. Mach. Learn.","author":"Clark","year":"2022"},{"key":"ref64","first-page":"18332","article-title":"DeepSpeed-MoE: Advancing mixture-of-experts inference and training to power next-generation AI scale","volume-title":"Proc. 39th Int. Conf. Mach. Learn.","author":"Rajbhandari","year":"2022"},{"key":"ref65","article-title":"Skywork-MoE: A deep dive into training techniques for mixture-of-experts language models","author":"Wei","year":"2024"},{"key":"ref66","article-title":"Jamba: A hybrid transformer-Mamba language model","author":"Lieber","year":"2024"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.70"},{"key":"ref68","article-title":"M6-T: Exploring sparse expert models and beyond","author":"Yang","year":"2021"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01138"},{"key":"ref70","article-title":"Yuan 2.0-M32: Mixture of experts with attention router","author":"Wu","year":"2024"},{"key":"ref71","article-title":"ModuleFormer: Learning modular large language models from uncurated data","author":"Shen","year":"2023"},{"key":"ref72","first-page":"6265","article-title":"Base layers: Simplifying training of large, sparse models","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Lewis","year":"2021"},{"key":"ref73","first-page":"29335","article-title":"DSelect-k: Differentiable selection in the mixture of experts with applications to multi-task learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Hazimeh","year":"2021"},{"key":"ref74","article-title":"Scalable and efficient MoE training for multitask multilingual models","author":"Kim","year":"2021"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.304"},{"key":"ref76","article-title":"No language left behind: Scaling human-centered machine translation","author":"Costa-juss\u00e0","year":"2022"},{"key":"ref77","first-page":"34600","article-title":"On the representation collapse of sparse mixture of experts","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chi","year":"2022"},{"key":"ref78","first-page":"2664","article-title":"Uni-perceiver-MoE: Learning sparse generalist models with conditional MoEs","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhu","year":"2022"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.489"},{"key":"ref80","first-page":"17555","article-title":"Hash layers for large sparse models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Roller","year":"2021"},{"key":"ref81","article-title":"Taming sparsely activated transformer with stochastic experts","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zuo","year":"2021"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.407"},{"issue":"107","key":"ref83","first-page":"1","article-title":"Beyond english-centric multilingual machine translation","volume":"22","author":"Fan","year":"2021","journal-title":"J. Mach. Learn. Res."},{"key":"ref84","article-title":"PanGu-$\\Sigma$\u03a3: Towards trillion parameter language model with sparse heterogeneous computing","author":"Ren","year":"2023"},{"key":"ref85","first-page":"7103","article-title":"Mixture-of-experts with expert choice routing","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhou","year":"2022"},{"key":"ref86","first-page":"42531","article-title":"Brainformers: Trading simplicity for efficiency","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","author":"Zhou","year":"2023"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/wacv61041.2025.00115"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.278"},{"key":"ref89","article-title":"JetMoE: Reaching Llama2 performance with 0.1 M dollars","author":"Shen","year":"2024"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.804"},{"key":"ref91","article-title":"MoE-LLaVA: Mixture of experts for large vision-language models","author":"Lin","year":"2024"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.280"},{"key":"ref93","article-title":"MixLoRA: Enhancing large language models fine-tuning with LoRA based mixture of experts","author":"Li","year":"2024"},{"key":"ref94","article-title":"LLaVA-MoLE: Sparse mixture of LoRA experts for mitigating data conflicts in instruction finetuning MLLMs","author":"Chen","year":"2024"},{"key":"ref95","article-title":"SiRA: Sparse mixture of low rank adaptation","author":"Zhu","year":"2023"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.433"},{"key":"ref97","article-title":"Higher layers need more LoRA experts","author":"Gao","year":"2024"},{"key":"ref98","article-title":"Intuition-aware mixture-of-rank-1-experts for parameter efficient finetuning","author":"Liu","year":"2024"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-32010-6_300155"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.244"},{"key":"ref101","article-title":"Experts weights averaging: A new general training scheme for vision transformers","author":"Huang","year":"2023"},{"key":"ref102","article-title":"Branch-train-Merge: Embarrassingly parallel training of expert language models","volume-title":"Proc. 1st Workshop Interpolation Regularizers Beyond NeurIPS","author":"Li","year":"2022"},{"key":"ref103","article-title":"Fusing models with complementary expertise","volume-title":"Proc. 12th Int. Conf. Learn. Representations","author":"Wang","year":"2023"},{"key":"ref104","article-title":"FastMoE: A fast mixture-of-expert training system","author":"He","year":"2021"},{"key":"ref105","first-page":"269","article-title":"Tutel: Adaptive mixture-of-experts at scale","volume-title":"Proc. Mach. Learn. Syst.","volume":"5","author":"Hwang","year":"2023"},{"key":"ref106","article-title":"SE-MoE: A scalable and efficient mixture-of-experts distributed training and inference system","author":"Shen","year":"2022"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.1145\/3503221.3508418"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1145\/3577193.3593704"},{"key":"ref109","article-title":"HetuMoE: An efficient trillion-scale mixture-of-expert distributed training system","author":"Nie","year":"2022"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.1145\/3588964"},{"key":"ref111","first-page":"961","article-title":"SmartMoE: Efficiently training Sparsely-Activated models through combining offline and online parallelization","volume-title":"Proc. USENIX Annu. Tech. Conf.","author":"Zhai","year":"2023"},{"key":"ref112","first-page":"288","article-title":"MegaBlocks: Efficient sparse training with mixture-of-experts","volume-title":"Proc. Mach. Learn. Syst.","volume":"5","author":"Gale","year":"2023"},{"key":"ref113","article-title":"Scattered mixture-of-experts implementation","author":"Tan","year":"2024"},{"key":"ref114","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613139"},{"key":"ref115","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS57955.2024.00086"},{"key":"ref116","first-page":"22173","article-title":"TA-MoE: Topology-aware large scale mixture-of-expert training","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chen","year":"2022"},{"key":"ref117","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2024.3385639"},{"key":"ref118","article-title":"Lancet: Accelerating mixture-of-experts training via whole graph computation-communication overlapping","author":"Jiang","year":"2024"},{"key":"ref119","article-title":"Shortcut-connected expert parallelism for accelerating mixture-of-experts","author":"Cai","year":"2024"},{"key":"ref120","doi-asserted-by":"publisher","DOI":"10.1109\/isca59077.2024.00078"},{"key":"ref121","article-title":"EdgeMoE: Fast on-device inference of MoE-based large language models","author":"Yi","year":"2023"},{"key":"ref122","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00015"},{"key":"ref123","first-page":"6074","article-title":"Patch-level routing in mixture-of-experts is provably sample-efficient for convolutional neural networks","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","author":"Chowdhury","year":"2023"},{"key":"ref124","doi-asserted-by":"publisher","DOI":"10.1145\/3383313.3412236"},{"key":"ref125","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3614773"},{"key":"ref126","first-page":"9564","article-title":"Multimodal contrastive learning with LiMoE: The language-image mixture of experts","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Mustafa","year":"2022"},{"key":"ref127","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.758"},{"key":"ref128","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2025.3532688"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73397-0_18"},{"key":"ref130","article-title":"Mistral 7B","author":"Jiang","year":"2023"},{"key":"ref131","article-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"key":"ref132","article-title":"ChatGPT: Optimizing language models for dialogue","year":"2022"},{"key":"ref133","article-title":"DeepSeek LLM: Scaling open-source language models with longtermism","author":"Bi","year":"2024"},{"key":"ref134","article-title":"Estimating or propagating gradients through stochastic neurons for conditional computation","author":"Bengio","year":"2013"},{"key":"ref135","article-title":"Low-rank approximations for conditional feedforward computation in deep neural networks","author":"Davis","year":"2013"},{"key":"ref136","first-page":"2549","article-title":"Dynamic capacity networks","volume-title":"Proc. 33rd Int. Conf. Mach. Learn.","author":"Almahairi","year":"2016"},{"key":"ref137","article-title":"Conditional computation in neural networks for faster models","author":"Bengio","year":"2015"},{"key":"ref138","article-title":"Routing networks: Adaptive selection of non-linear functions for multi-task learning","author":"Rosenbaum","year":"2017"},{"key":"ref139","article-title":"Routing networks and the challenges of modular and compositional computation","author":"Rosenbaum","year":"2019"},{"key":"ref140","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.361"},{"key":"ref141","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-emnlp.189"},{"key":"ref142","article-title":"OLMoE: Open mixture-of-experts language models","author":"Muennighoff","year":"2024"},{"key":"ref143","article-title":"Qwen1.5-MoE: Matching 7B model performance with 1\/3 activated parameters","year":"2024"},{"key":"ref144","article-title":"Measuring massive multitask language understanding","author":"Hendrycks","year":"2020"},{"key":"ref145","article-title":"Training verifiers to solve math word problems","author":"Cobbe","year":"2021"},{"key":"ref146","article-title":"Measuring mathematical problem solving with the math dataset","author":"Hendrycks","year":"2021"},{"key":"ref147","article-title":"Evaluating large language models trained on code","author":"Chen","year":"2021"},{"key":"ref148","article-title":"Mamba: Linear-time sequence modeling with selective state spaces","author":"Gu","year":"2023"},{"key":"ref149","doi-asserted-by":"crossref","DOI":"10.21203\/rs.3.rs-1553541\/v1","article-title":"Delta tuning: A comprehensive study of parameter efficient methods for pre-trained language models","author":"Ding","year":"2022"},{"key":"ref150","article-title":"Parameter-efficient fine-tuning for large models: A comprehensive survey","author":"Han","year":"2024"},{"key":"ref151","article-title":"Scaling down to scale up: A guide to parameter-efficient fine-tuning","author":"Lialin","year":"2023"},{"key":"ref152","article-title":"MOELoRA: An MoE-based parameter efficient fine-tuning method for multi-task medical applications","author":"Liu","year":"2023"},{"key":"ref153","article-title":"A case study of instruction tuning with mixture of parameter-efficient experts","volume-title":"Proc. NeurIPS 2023 Workshop Instruct. Tuning Instruct. Following","author":"Ostapenko","year":"2023"},{"key":"ref154","article-title":"LoRA: Low-rank adaptation of large language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Hu","year":"2021"},{"key":"ref155","first-page":"2790","article-title":"Parameter-efficient transfer learning for NLP","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Houlsby","year":"2019"},{"key":"ref156","first-page":"1950","article-title":"Few-shot parameter-efficient fine-tuning is better and cheaper than in-context learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Liu","year":"2022"},{"key":"ref157","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acllong.353"},{"key":"ref158","article-title":"Skywork: A more open bilingual foundation model","author":"Wei","year":"2023"},{"key":"ref159","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.116"},{"key":"ref160","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"ref161","first-page":"551","article-title":"ZeRO-offload: Democratizing billion-scale model training","volume-title":"Proc. 2021 USENIX Annu. Tech. Conf.","author":"Ren","year":"2021"},{"key":"ref162","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476205"},{"key":"ref163","doi-asserted-by":"publisher","DOI":"10.1145\/3503221.3508417"},{"key":"ref164","first-page":"559","article-title":"Alpa: Automating inter-and Intra-Operator parallelism for distributed deep learning","volume-title":"Proc. 16th USENIX Symp. Operating Syst. Des. Implementation","author":"Zheng","year":"2022"},{"key":"ref165","article-title":"Megatron-LM: Training multi-billion parameter language models using model parallelism","author":"Shoeybi","year":"2019"},{"key":"ref166","article-title":"Using DeepSpeed and megatron to train megatron-turing NLG 530B, a large-scale generative language model","author":"Smith","year":"2022"},{"key":"ref167","first-page":"1","article-title":"Efficient large-scale language model training on GPU clusters using Megatron-LM","volume-title":"Proc. Int. Conf. High Perform. Comput. Netw., Storage Anal.","author":"Narayanan","year":"2021"},{"key":"ref168","doi-asserted-by":"publisher","DOI":"10.48550\/arxiv.1811.06965"},{"key":"ref169","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"ref170","article-title":"Zero bubble pipeline parallelism","volume-title":"Proc. 12th Int. Conf. Learn. Representations","author":"Qi","year":"2023"},{"key":"ref171","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.134"},{"key":"ref172","first-page":"341","article-title":"Reducing activation recomputation in large transformer models","volume-title":"Proc. Mach. Learn. Syst.","volume":"5","author":"Korthikanti","year":"2023"},{"key":"ref173","doi-asserted-by":"publisher","DOI":"10.1145\/3662158.3662806"},{"key":"ref174","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-4009"},{"key":"ref175","first-page":"10435","article-title":"Mesh-TensorFlow: Deep learning for supercomputers","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Shazeer","year":"2018"},{"key":"ref176","article-title":"Introducing Qwen1.5","year":"2024"},{"key":"ref177","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"ref178","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2021.11.041"},{"key":"ref179","first-page":"689","article-title":"Multimodal deep learning","volume-title":"Proc. 28th Int. Conf. Mach. Learn.","author":"Ngiam","year":"2011"},{"key":"ref180","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2798607"},{"key":"ref181","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2021.07.009"},{"key":"ref182","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.7005"}],"container-title":["IEEE Transactions on Knowledge and Data Engineering"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/69\/4358933\/10937907.pdf?arnumber=10937907","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,6]],"date-time":"2025-06-06T00:19:34Z","timestamp":1749169174000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10937907\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":182,"URL":"https:\/\/doi.org\/10.1109\/tkde.2025.3554028","relation":{},"ISSN":["1041-4347","1558-2191","2326-3865"],"issn-type":[{"value":"1041-4347","type":"print"},{"value":"1558-2191","type":"electronic"},{"value":"2326-3865","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]}}}