{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T17:13:32Z","timestamp":1780420412113,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T00:00:00Z","timestamp":1781913600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"funder":[{"name":"Bundesministerium f\u00fcr Forschung, Technologie und Raumfahrt (BMFTR, German Federal Ministry of Research, Technology and Space)","award":["01IS23068"],"award-info":[{"award-number":["01IS23068"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,21]]},"DOI":"10.1145\/3812836.3814785","type":"proceedings-article","created":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T16:54:53Z","timestamp":1780419293000},"page":"348-353","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["FaaSMoE: A Serverless Framework for Multi-Tenant Mixture-of-Experts Serving"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-3780-5828","authenticated-orcid":false,"given":"Minghe","family":"Wang","sequence":"first","affiliation":[{"name":"TU Berlin, Berlin, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9277-3032","authenticated-orcid":false,"given":"Trever","family":"Schirmer","sequence":"additional","affiliation":[{"name":"TU Berlin, Berlin, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-0627-3600","authenticated-orcid":false,"given":"Mohammadreza","family":"Malekabbasi","sequence":"additional","affiliation":[{"name":"TU Berlin, Berlin, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7524-3256","authenticated-orcid":false,"given":"David","family":"Bermbach","sequence":"additional","affiliation":[{"name":"TU Berlin, Berlin, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,20]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Amazon Web Services. [n. d.]. AWS Lambda Pricing. https:\/\/aws.amazon.com\/lambda\/pricing\/."},{"key":"e_1_3_2_1_2_1","volume-title":"Serverless Computing: Current Trends and Open Problems. In Research Advances in Cloud Computing","author":"Baldini Ioana","year":"2017","unstructured":"Ioana Baldini, Paul Castro, Kerry Chang, Perry Cheng, Stephen Fink, Vatche Ishakian, Nick Mitchel, Vinod Muthusamy, Rodric Rabbah, Aleksander Slominski, and Philippe Suter. 2017. Serverless Computing: Current Trends and Open Problems. In Research Advances in Cloud Computing. Springer."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/IC2E52221.2021.00044"},{"key":"e_1_3_2_1_4_1","volume-title":"Deepseekmoe: Towards ultimate expert specialization in mixture-of-experts language models. arXiv preprint arXiv:2401.06066","author":"Dai Damai","year":"2024","unstructured":"Damai Dai, Chengqi Deng, Chenggang Zhao, RX Xu, Huazuo Gao, Deli Chen, Jiashi Li, Wangding Zeng, Xingkai Yu, Yu Wu, et al. 2024. Deepseekmoe: Towards ultimate expert specialization in mixture-of-experts language models. arXiv preprint arXiv:2401.06066 (2024)."},{"key":"e_1_3_2_1_5_1","unstructured":"Databricks on AWS. 2025. Serverless GPU Compute on Databricks. https:\/\/docs.databricks.com\/aws\/en\/compute\/serverless\/gpu."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3698038.3698521"},{"key":"e_1_3_2_1_7_1","volume-title":"International conference on machine learning. PMLR, 5547\u20135569","author":"Du Nan","year":"2022","unstructured":"Nan Du, Yanping Huang, Andrew M Dai, Simon Tong, Dmitry Lepikhin, Yuanzhong Xu, Maxim Krikun, Yanqi Zhou, Adams Wei Yu, Orhan Firat, et al. 2022. Glam: Efficient scaling of language models with mixture-of-experts. In International conference on machine learning. PMLR, 5547\u20135569."},{"key":"e_1_3_2_1_8_1","first-page":"1","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus William","year":"2022","unstructured":"William Fedus, Barret Zoph, and Noam Shazeer. 2022. Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity. Journal of Machine Learning Research 23, 120 (2022), 1\u201339.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_9_1","volume-title":"Efficient CPU-GPU Collaborative Inference for MoE-based LLMs on Memory-Limited Systems. arXiv preprint arXiv:2512.16473","author":"Huang En-Ming","year":"2025","unstructured":"En-Ming Huang, Li-Shang Lin, and Chun-Yi Lee. 2025. Efficient CPU-GPU Collaborative Inference for MoE-based LLMs on Memory-Limited Systems. arXiv preprint arXiv:2512.16473 (2025)."},{"key":"e_1_3_2_1_10_1","volume-title":"Diego de las Casas, Emma Bou Hanna, Florian Bressand, et al.","author":"Jiang Albert Q","year":"2024","unstructured":"Albert Q Jiang, Alexandre Sablayrolles, Antoine Roux, Arthur Mensch, Blanche Savary, Chris Bamford, Devendra Singh Chaplot, Diego de las Casas, Emma Bou Hanna, Florian Bressand, et al. 2024. Mixtral of experts. arXiv preprint arXiv:2401.04088 (2024)."},{"key":"e_1_3_2_1_11_1","volume-title":"Gshard: Scaling giant models with conditional computation and automatic sharding. arXiv preprint arXiv:2006.16668","author":"Lepikhin Dmitry","year":"2020","unstructured":"Dmitry Lepikhin, HyoukJoong Lee, Yuanzhong Xu, Dehao Chen, Orhan Firat, Yanping Huang, Maxim Krikun, Noam Shazeer, and Zhifeng Chen. 2020. Gshard: Scaling giant models with conditional computation and automatic sharding. arXiv preprint arXiv:2006.16668 (2020)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM55648.2025.11044553"},{"key":"e_1_3_2_1_13_1","unstructured":"Ziming Liu Boyu Tian Guoteng Wang Zhen Jiang Peng Sun Zhenhua Han Tian Tang Xiaohe Hu Yanmin Jia Yan Zhang et al. 2025. Expert-as-a-Service: Towards Efficient Scalable and Robust Large-scale MoE Serving. arXiv preprint arXiv:2509.17863 (2025)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICFC49376.2020.00011"},{"key":"e_1_3_2_1_15_1","volume-title":"International conference on machine learning. PMLR","author":"Rajbhandari Samyam","year":"2022","unstructured":"Samyam Rajbhandari, Conglong Li, Zhewei Yao, Minjia Zhang, Reza Yazdani Aminabadi, Ammar Ahmad Awan, Jeff Rasley, and Yuxiong He. 2022. Deepspeed-moe: Advancing mixture-of-experts inference and training to power next-generation ai scale. In International conference on machine learning. PMLR, 18332\u201318346."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476205"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406703"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/IC2E65552.2025.00017"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3592533.3592808"},{"key":"e_1_3_2_1_20_1","unstructured":"Mohammad Shahrad Rodrigo Fonseca Inigo Goiri Gohar Chaudhry Paul Batum Jason Cooke Eduardo Laureano Colby Tresness Mark Russinovich and Ricardo Bianchini. 2020. Serverless in the wild: Characterizing and optimizing the serverless workload at a large cloud provider. In 2020 USENIX annual technical conference (USENIX ATC 20). 205\u2013218."},{"key":"e_1_3_2_1_21_1","volume-title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer. arXiv preprint arXiv:1701.06538","author":"Shazeer Noam","year":"2017","unstructured":"Noam Shazeer, Azalia Mirhoseini, Krzysztof Maziarz, Andy Davis, Quoc Le, Geoffrey Hinton, and Jeff Dean. 2017. Outrageously large neural networks: The sparsely-gated mixture-of-experts layer. arXiv preprint arXiv:1701.06538 (2017)."},{"key":"e_1_3_2_1_22_1","unstructured":"Qwen Team et al. 2024. Qwen2 technical report. arXiv preprint arXiv:2407.10671 2 3 (2024)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3578354.3592869"},{"key":"e_1_3_2_1_24_1","volume-title":"WDMoE: Wireless distributed mixture of experts for large language models","author":"Xue Nan","year":"2025","unstructured":"Nan Xue, Yaping Sun, Zhiyong Chen, Meixia Tao, Xiaodong Xu, Liang Qian, Shuguang Cui, Wenjun Zhang, and Ping Zhang. 2025. WDMoE: Wireless distributed mixture of experts for large language models. IEEE Transactions on Wireless Communications (2025)."},{"key":"e_1_3_2_1_25_1","volume-title":"fMoE: Fine-Grained Expert Offloading for Large Mixture-of-Experts Serving. arXiv preprint arXiv:2502.05370","author":"Yu Hanfei","year":"2025","unstructured":"Hanfei Yu, Xingqi Cui, Hong Zhang, and Hao Wang. 2025. fMoE: Fine-Grained Expert Offloading for Large Mixture-of-Experts Serving. arXiv preprint arXiv:2502.05370 (2025)."},{"key":"e_1_3_2_1_26_1","volume-title":"FloE: On-the-Fly MoE Inference on Memory-constrained GPU. arXiv preprint arXiv:2505.05950","author":"Zhou Yuxin","year":"2025","unstructured":"Yuxin Zhou, Zheng Li, Jun Zhang, Jue Wang, Yiping Wang, Zhongle Xie, Ke Chen, and Lidan Shou. 2025. FloE: On-the-Fly MoE Inference on Memory-constrained GPU. arXiv preprint arXiv:2505.05950 (2025)."}],"event":{"name":"MobiSys Workshop '26: 24th Annual International Conference on Mobile Systems, Applications and Services Workshops","location":"University of Cambridge Cambridge United Kingdom","acronym":"MobiSys Workshop '26","sponsor":["SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing","SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 24th Annual International Conference on Mobile Systems, Applications and Services Workshops"],"original-title":[],"deposited":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T16:56:53Z","timestamp":1780419413000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3812836.3814785"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,20]]},"references-count":26,"alternative-id":["10.1145\/3812836.3814785","10.1145\/3812836"],"URL":"https:\/\/doi.org\/10.1145\/3812836.3814785","relation":{},"subject":[],"published":{"date-parts":[[2026,6,20]]},"assertion":[{"value":"2026-06-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}