{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T15:44:09Z","timestamp":1782834249892,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","funder":[{"name":"Hong Kong RGC General Research Fund","award":["152244\/21E"],"award-info":[{"award-number":["152244\/21E"]}]},{"name":"Hong Kong RGC General Research Fund","award":["152169\/22E"],"award-info":[{"award-number":["152169\/22E"]}]},{"name":"Hong Kong RGC General Research Fund","award":["152228\/23E"],"award-info":[{"award-number":["152228\/23E"]}]},{"name":"Hong Kong RGC General Research Fund","award":["162161\/24E"],"award-info":[{"award-number":["162161\/24E"]}]},{"name":"Research Impact Fund","award":["R5011-23F"],"award-info":[{"award-number":["R5011-23F"]}]},{"name":"Research Impact Fund","award":["R5060-19"],"award-info":[{"award-number":["R5060-19"]}]},{"name":"Collaborative Research Fund","award":["C1042-23GF"],"award-info":[{"award-number":["C1042-23GF"]}]},{"name":"NSFC\/RGC Collaborative Research Scheme","award":["CRS-HKUST602\/24"],"award-info":[{"award-number":["CRS-HKUST602\/24"]}]},{"name":"Areas of Excellence Scheme","award":["AoE\/E-601\/22-R"],"award-info":[{"award-number":["AoE\/E-601\/22-R"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,3]]},"DOI":"10.1145\/3680207.3723493","type":"proceedings-article","created":{"date-parts":[[2025,11,21]],"date-time":"2025-11-21T13:19:18Z","timestamp":1763731158000},"page":"574-588","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["D2MoE: Dual Routing and Dynamic Scheduling for Efficient On-Device MoE-based LLM Serving"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-8977-850X","authenticated-orcid":false,"given":"Haodong","family":"Wang","sequence":"first","affiliation":[{"name":"Hong Kong University of Science and Technology, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0328-2894","authenticated-orcid":false,"given":"Qihua","family":"Zhou","sequence":"additional","affiliation":[{"name":"College of Computer Science and Software Engineering, Shenzhen University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5689-382X","authenticated-orcid":false,"given":"Zicong","family":"Hong","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9831-2202","authenticated-orcid":false,"given":"Song","family":"Guo","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,11,21]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the AAAI conference on artificial intelligence (AAAI '22","volume":"7439","author":"Bisk Yonatan","year":"2020","unstructured":"Yonatan Bisk, Rowan Zellers, Ronan Le Bras, Jianfeng Gao, and Yejin Choi. 2020. PIQA: Reasoning about Physical Commonsense in Natural Language. In Proceedings of the AAAI conference on artificial intelligence (AAAI '22, Vol. 34). 7432\u20137439."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3495883"},{"key":"e_1_3_2_1_3_1","unstructured":"Shiyi Cao Shu Liu Tyler Griggs Peter Schafhalter Xiaoxuan Liu Ying Sheng Joseph E. Gonzalez Matei Zaharia and Ion Stoica. 2025. MoE-Lightning: High-Throughput MoE Inference on Memory-constrained GPUs."},{"key":"e_1_3_2_1_4_1","volume-title":"IEEE Conference on Computer Communications (INFOCOM '24)","author":"Chen Jinyu","year":"2024","unstructured":"Jinyu Chen, Wenchao Xu, Zicong Hong, Song Guo, Haozhao Wang, Jie Zhang, and Deze Zeng. 2024. OTAS: An Elastic Transformer Serving System via Token Adaptation. In IEEE Conference on Computer Communications (INFOCOM '24). 1021\u20131030."},{"key":"e_1_3_2_1_5_1","unstructured":"Mark Chen Jerry Tworek Heewoo Jun Qiming Yuan Henrique Ponde Jared Kaplan Harrison Edwards Yura Burda Nicholas Joseph Greg Brockman Alex Ray Raul Puri Gretchen Krueger Michael Petrov Heidy Khlaaf Girish Sastry Pamela Mishkin Brooke Chan Scott Gray Nick Ryder Mikhail Pavlov Alethea Power Lukasz Kaiser Mohammad Bavarian Clemens Winter Philippe Tillet Felipe Petroski Such David W. Cummings Matthias Plappert Fotios Chantzis Elizabeth Barnes Ariel Herbert-Voss William H. Guss Alex Nichol Igor Babuschkin Suchir Balaji Shantanu Jain Andrew Carr Jan Leike Joshua Achiam Vedant Misra Evan Morikawa Alec Radford Matthew M. Knight Miles Brundage Mira Murati Katie Mayer Peter Welinder Bob McGrew Dario Amodei Sam McCandlish Ilya Sutskever and Wojciech Zaremba. 2021. Evaluating Large Language Models Trained on Code. ArXiv abs\/2107.03374 (2021)."},{"key":"e_1_3_2_1_6_1","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","volume":"1","author":"Clark Christopher","year":"2019","unstructured":"Christopher Clark, Kenton Lee, Ming-Wei Chang, Tom Kwiatkowski, Michael Collins, and Kristina Toutanova. 2019. BoolQ: Exploring the Surprising Difficulty of Natural Yes\/No Questions. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Vol. 1. 2924\u20132936."},{"key":"e_1_3_2_1_7_1","unstructured":"Peter Clark Isaac Cowhey Oren Etzioni Tushar Khot Ashish Sabharwal Carissa Schoenick and Oyvind Tafjord. 2018. Think you have Solved Question Answering? Try ARC the AI2 Reasoning Challenge. arXiv:1803.05457"},{"key":"e_1_3_2_1_8_1","first-page":"1","article-title":"Switch transformers: scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus William","year":"2022","unstructured":"William Fedus, Barret Zoph, and Noam Shazeer. 2022. Switch transformers: scaling to trillion parameter models with simple and efficient sparsity. The Journal of Machine Learning Research 23, 1 (jan 2022), 39 pages.","journal-title":"The Journal of Machine Learning Research"},{"key":"e_1_3_2_1_9_1","volume-title":"GPTQ: Accurate Post-training Compression for Generative Pretrained Transformers. The Eleventh International Conference on Learning Representations","author":"Frantar Elias","year":"2023","unstructured":"Elias Frantar, Saleh Ashkboos, Torsten Hoefler, and Dan Alistarh. 2023. GPTQ: Accurate Post-training Compression for Generative Pretrained Transformers. The Eleventh International Conference on Learning Representations (2023)."},{"key":"e_1_3_2_1_10_1","volume-title":"Haonan Li, Kyle McDonell, Niklas Muennighoff, Chris Ociepa, Jason Phang, Laria Reynolds, Hailey Schoelkopf, Aviya Skowron, Lintang Sutawika, Eric Tang, Anish Thite, Ben Wang, Kevin Wang, and Andy Zou.","author":"Gao Leo","year":"2024","unstructured":"Leo Gao, Jonathan Tow, Baber Abbasi, Stella Biderman, Sid Black, Anthony DiPofi, Charles Foster, Laurence Golding, Jeffrey Hsu, Alain Le Noac'h, Haonan Li, Kyle McDonell, Niklas Muennighoff, Chris Ociepa, Jason Phang, Laria Reynolds, Hailey Schoelkopf, Aviya Skowron, Lintang Sutawika, Eric Tang, Anish Thite, Ben Wang, Kevin Wang, and Andy Zou. 2024. A framework for few-shot language model evaluation."},{"key":"e_1_3_2_1_11_1","unstructured":"Georgi Gerganov. 2023. llama.cpp. https:\/\/github.com\/ggerganov\/llama.cpp"},{"key":"e_1_3_2_1_12_1","unstructured":"Github. 2022. Copilot. https:\/\/github.com\/features\/copilot"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing. 20924\u201320938","author":"Gong Zhuocheng","year":"2024","unstructured":"Zhuocheng Gong, Ang Lv, Jian Guan, Wei Wu, Huishuai Zhang, Minlie Huang, Dongyan Zhao, and Rui Yan. 2024. Mixture-of-Modules: Reinventing Transformers as Dynamic Assemblies of Modules. In Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing. 20924\u201320938."},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS","author":"Guo Liwei","year":"2023","unstructured":"Liwei Guo, Wonkyo Choe, and Felix Xiaozhu Lin. 2023. STI: Turbocharge NLP Inference at the Edge via Elastic Pipelining. In Proceedings of the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS 2023). 791\u2013803."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3492321.3519565"},{"key":"e_1_3_2_1_16_1","volume-title":"The Eleventh International Conference on Learning Representations","author":"Huang Wei","year":"2025","unstructured":"Wei Huang, Yue Liao, Jianhui Liu, Ruifei He, Haoru Tan, Shiming Zhang, Hongsheng Li, Si Liu, and Xiaojuan Qi. 2025. Mc-moe: Mixture compressor for mixture-of-experts llms gains more. The Eleventh International Conference on Learning Representations (2025)."},{"key":"e_1_3_2_1_17_1","unstructured":"Albert Q. Jiang Alexandre Sablayrolles Antoine Roux Arthur Mensch Blanche Savary Chris Bamford Devendra Singh Chaplot Diego de las Casas Emma Bou Hanna Florian Bressand Gianna Lengyel Guillaume Bour Guillaume Lample L\u00e9lio Renard Lavaud Lucile Saulnier Marie-Anne Lachaux Pierre Stock Sandeep Subramanian Sophia Yang Szymon Antoniak Teven Le Scao Th\u00e9ophile Gervet Thibaut Lavril Thomas Wang Timoth\u00e9e Lacroix and William El Sayed. 2024. Mixtral of Experts. arXiv:2401.04088"},{"key":"e_1_3_2_1_18_1","volume-title":"Fiddler: CPU-GPU Orchestration for Fast Inference of Mixture-of-Experts Models. In 5th Workshop on practical ML for limited\/low resource settings.","author":"Kamahori Keisuke","year":"2024","unstructured":"Keisuke Kamahori, Yile Gu, Kan Zhu, and Baris Kasikci. 2024. Fiddler: CPU-GPU Orchestration for Fast Inference of Mixture-of-Experts Models. In 5th Workshop on practical ML for limited\/low resource settings."},{"key":"e_1_3_2_1_19_1","unstructured":"Young Jin Kim Raffy Fahim and Hany Hassan Awadalla. 2023. Mixture of Quantized Experts (MoQE): Complementary Effect of Low-bit Quantization and Robustness. arXiv:2310.02410"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3636534.3649391"},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of Machine Learning and Systems (MLSys '24","volume":"100","author":"Lin Ji","year":"2024","unstructured":"Ji Lin, Jiaming Tang, Haotian Tang, Shang Yang, Wei-Ming Chen, WeiChen Wang, Guangxuan Xiao, Xingyu Dang, Chuang Gan, and Song Han. 2024. AWQ: Activation-aware Weight Quantization for LLM Compression and Acceleration. In Proceedings of Machine Learning and Systems (MLSys '24, Vol. 6). 87\u2013100."},{"key":"e_1_3_2_1_23_1","first-page":"87","article-title":"AWQ: Activation-aware Weight Quantization for On-Device LLM Compression and Acceleration","volume":"6","author":"Lin Ji","year":"2024","unstructured":"Ji Lin, Jiaming Tang, Haotian Tang, Shang Yang, Wei-Ming Chen, Wei-Chen Wang, Guangxuan Xiao, Xingyu Dang, Chuang Gan, and Song Han. 2024. AWQ: Activation-aware Weight Quantization for On-Device LLM Compression and Acceleration. Proceedings of Machine Learning and Systems 6 (2024), 87\u2013100.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_24_1","volume-title":"QServe: W4A8KV4 Quantization and System Co-design for Efficient LLM Serving. arXiv preprint arXiv:2405.04532","author":"Yujun","year":"2024","unstructured":"Yujun Lin*, Haotian Tang*, Shang Yang*, Zhekai Zhang, Guangxuan Xiao, Chuang Gan, and Song Han. 2024. QServe: W4A8KV4 Quantization and System Co-design for Efficient LLM Serving. arXiv preprint arXiv:2405.04532 (2024)."},{"key":"e_1_3_2_1_25_1","volume-title":"AffineQuant: Affine Transformation Quantization for Large Language Models. In The Twelfth International Conference on Learning Representations (ICLR '24)","author":"Ma Yuexiao","year":"2024","unstructured":"Yuexiao Ma, Huixia Li, Xiawu Zheng, Feng Ling, Xuefeng Xiao, Rui Wang, Shilei Wen, Fei Chao, and Rongrong Ji. 2024. AffineQuant: Affine Transformation Quantization for Large Language Models. In The Twelfth International Conference on Learning Representations (ICLR '24)."},{"key":"e_1_3_2_1_26_1","unstructured":"Stephen Merity Caiming Xiong James Bradbury and Richard Socher. 2016. Pointer Sentinel Mixture Models. arXiv:1609.07843"},{"key":"e_1_3_2_1_27_1","unstructured":"NVIDIA. 2023b. NVIDIA. Tensorrt-llm. https:\/\/github.com\/NVIDIA\/TensorRT-LLM"},{"key":"e_1_3_2_1_28_1","unstructured":"OpenAI. 2022. Chatgpt. https:\/\/openai.com\/blog\/chatgpt"},{"key":"e_1_3_2_1_29_1","volume-title":"Different-Sized LLMs. In Proceedings of the 41st International Conference on Machine Learning (ICML '24)","author":"Park Yeonhong","unstructured":"Yeonhong Park, Jake Hyun, SangLyul Cho, Bonggeun Sim, and Jae W. Lee. 2024. Any-Precision LLM: Low-Cost Deployment of Multiple, Different-Sized LLMs. In Proceedings of the 41st International Conference on Machine Learning (ICML '24)."},{"key":"e_1_3_2_1_30_1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter J. Liu. 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. Journal of Machine Learning Research 21, 1, Article 140 (2020), 67 pages.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_31_1","volume-title":"Peter Conway Humphreys, and Adam Santoro","author":"Raposo David","year":"2024","unstructured":"David Raposo, Sam Ritter, Blake Richards, Timothy Lillicrap, Peter Conway Humphreys, and Adam Santoro. 2024. Mixture-of-Depths: Dynamically allocating compute in transformer-based language models. arXiv:2404.02258"},{"key":"e_1_3_2_1_32_1","volume-title":"Peter Conway Humphreys, and Adam Santoro","author":"Raposo David","year":"2024","unstructured":"David Raposo, Sam Ritter, Blake Richards, Timothy Lillicrap, Peter Conway Humphreys, and Adam Santoro. 2024. Mixture-of-Depths: Dynamically allocating compute in transformer-based language models. arXiv:2404.02258"},{"key":"e_1_3_2_1_33_1","first-page":"9","article-title":"WinoGrande: an adversarial winograd schema challenge at scale","volume":"64","author":"Sakaguchi Keisuke","year":"2021","unstructured":"Keisuke Sakaguchi, Ronan Le Bras, Chandra Bhagavatula, and Yejin Choi. 2021. WinoGrande: an adversarial winograd schema challenge at scale. Commun. ACM 64, 9 (aug 2021), 99\u2013106.","journal-title":"Commun. ACM"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","unstructured":"Yixin Song Zeyu Mi Haotong Xie and Haibo Chen. 2023. PowerInfer: Fast Large Language Model Serving with a Consumer-grade GPU. arXiv:2312.12456","DOI":"10.1145\/3694715.3695964"},{"key":"e_1_3_2_1_35_1","volume-title":"Llumnix: Dynamic Scheduling for Large Language Model Serving. 18th USENIX Symposium on Operating Systems Design and Implementation","author":"Sun Biao","year":"2024","unstructured":"Biao Sun, Ziming Huang, Hanyu Zhao, Wencong Xiao, Xinyi Zhang, Yong Li, and Wei Lin. 2024. Llumnix: Dynamic Scheduling for Large Language Model Serving. 18th USENIX Symposium on Operating Systems Design and Implementation (2024)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3315508.3329973"},{"key":"e_1_3_2_1_37_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale Dan Bikel Lukas Blecher Cristian Canton Ferrer Moya Chen Guillem Cucurull David Esiobu Jude Fernandes Jeremy Fu Wenyin Fu Brian Fuller Cynthia Gao Vedanuj Goswami Naman Goyal Anthony Hartshorn Saghar Hosseini Rui Hou Hakan Inan Marcin Kardas Viktor Kerkez Madian Khabsa Isabel Kloumann Artem Korenev Punit Singh Koura Marie-Anne Lachaux Thibaut Lavril Jenya Lee Diana Liskovich Yinghai Lu Yuning Mao Xavier Martinet Todor Mihaylov Pushkar Mishra Igor Molybog Yixin Nie Andrew Poulton Jeremy Reizenstein Rashi Rungta Kalyan Saladi Alan Schelten Ruan Silva Eric Michael Smith Ranjan Subramanian Xiaoqing Ellen Tan Binh Tang Ross Taylor Adina Williams Jian Xiang Kuan Puxin Xu Zheng Yan Iliyan Zarov Yuchen Zhang Angela Fan Melanie Kambadur Sharan Narang Aurelien Rodriguez Robert Stojnic Sergey Edunov and Thomas Scialom. 2023. Llama 2: Open Foundation and Fine-Tuned Chat Models. arXiv:2307.09288"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295349"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447993.3448625"},{"key":"e_1_3_2_1_40_1","volume-title":"2024 USENIX Annual Technical Conference (USENIX ATC '24)","author":"Xia Haojun","year":"2024","unstructured":"Haojun Xia, Zhen Zheng, Xiaoxia Wu, Shiyang Chen, Zhewei Yao, Stephen Youn, Arash Bakhtiari, Michael Wyatt, Donglin Zhuang, Zhongzhu Zhou, Olatunji Ruwase, Yuxiong He, and Shuaiwen Leon Song. 2024. Quant-LLM: Accelerating the Serving of Large Language Models via FP6-Centric Algorithm-System Co-Design on Modern GPUs. In 2024 USENIX Annual Technical Conference (USENIX ATC '24). 699\u2013713."},{"key":"e_1_3_2_1_41_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning (ICML '23)","author":"Xiao Guangxuan","year":"2023","unstructured":"Guangxuan Xiao, Ji Lin, Mickael Seznec, Hao Wu, Julien Demouth, and Song Han. 2023. SmoothQuant: Accurate and Efficient Post-Training Quantization for Large Language Models. In Proceedings of the 40th International Conference on Machine Learning (ICML '23)."},{"key":"e_1_3_2_1_42_1","volume-title":"EdgeMoE: Fast On-Device Inference of MoE-based Large Language Models. ArXiv abs\/2308.14352","author":"Yi Rongjie","year":"2023","unstructured":"Rongjie Yi, Liwei Guo, Shiyun Wei, Ao Zhou, Shangguang Wang, and Mengwei Xu. 2023. EdgeMoE: Fast On-Device Inference of MoE-based Large Language Models. ArXiv abs\/2308.14352 (2023)."},{"key":"e_1_3_2_1_43_1","volume-title":"Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics (ACL '19)","author":"Zellers Rowan","year":"2019","unstructured":"Rowan Zellers, Ari Holtzman, Yonatan Bisk, Ali Farhadi, and Yejin Choi. 2019. HellaSwag: Can a Machine Really Finish Your Sentence?. In Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics (ACL '19). 4791\u20134800."},{"key":"e_1_3_2_1_44_1","volume-title":"Todor Mihaylov, Myle Ott, Sam Shleifer, Kurt Shuster, Daniel Simig, Punit Singh Koura, Anjali Sridhar, Tianlu Wang, and Luke Zettlemoyer.","author":"Zhang Susan","year":"2022","unstructured":"Susan Zhang, Stephen Roller, Naman Goyal, Mikel Artetxe, Moya Chen, Shuohui Chen, Christopher Dewan, Mona Diab, Xian Li, Xi Victoria Lin, Todor Mihaylov, Myle Ott, Sam Shleifer, Kurt Shuster, Daniel Simig, Punit Singh Koura, Anjali Sridhar, Tianlu Wang, and Luke Zettlemoyer. 2022. OPT: Open Pre-trained Transformer Language Models. arXiv:2205.01068"},{"key":"e_1_3_2_1_45_1","volume-title":"IEEE International Conference on Computer-Aided Design (ICCAD '24)","author":"Zhong Shuzhang","year":"2024","unstructured":"Shuzhang Zhong, Ling Liang, Yuan Wang, Runsheng Wang, and Meng Li Ru Huang. 2024. AdapMoE: Adaptive Sensitivity-based Expert Gating and Management for Efficient MoE Inference.. In IEEE International Conference on Computer-Aided Design (ICCAD '24)."},{"key":"e_1_3_2_1_46_1","volume-title":"LLaMA-MoE: Building Mixture-of-Experts from LLaMA with Continual Pre-training. arXiv preprint arXiv:2406.16554","author":"Zhu Tong","year":"2024","unstructured":"Tong Zhu, Xiaoye Qu, Daize Dong, Jiacheng Ruan, Jingqi Tong, Conghui He, and Yu Cheng. 2024. LLaMA-MoE: Building Mixture-of-Experts from LLaMA with Continual Pre-training. arXiv preprint arXiv:2406.16554 (2024)."}],"event":{"name":"ACM MOBICOM '25: 31st Annual International Conference on Mobile Computing and Networking","location":"Kerry Hotel, Hong Kong Hong Kong China","acronym":"ACM MOBICOM '25","sponsor":["SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing"]},"container-title":["Proceedings of the 31st Annual International Conference on Mobile Computing and Networking"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3680207.3723493","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,21]],"date-time":"2025-11-21T13:19:31Z","timestamp":1763731171000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3680207.3723493"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,3]]},"references-count":46,"alternative-id":["10.1145\/3680207.3723493","10.1145\/3680207"],"URL":"https:\/\/doi.org\/10.1145\/3680207.3723493","relation":{},"subject":[],"published":{"date-parts":[[2025,11,3]]},"assertion":[{"value":"2025-11-21","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}