{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T15:31:43Z","timestamp":1773588703326,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":109,"publisher":"ACM","funder":[{"name":"The National Natural Science Foundation of China","award":["U25B2021"],"award-info":[{"award-number":["U25B2021"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,3,22]]},"DOI":"10.1145\/3779212.3790168","type":"proceedings-article","created":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T13:55:26Z","timestamp":1773150926000},"page":"838-859","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["gShare: Efficient GPU Sharing with Aggressive Scheduling in Multi-tenant FaaS platform"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2222-393X","authenticated-orcid":false,"given":"Yanan","family":"Yang","sequence":"first","affiliation":[{"name":"China Telecom Cloud Computing Research Institute, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-8825-6245","authenticated-orcid":false,"given":"Zhengxiong","family":"Jiang","sequence":"additional","affiliation":[{"name":"China Telecom Cloud Technology Co. Ltd., Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2578-8660","authenticated-orcid":false,"given":"Meiqi","family":"Zhu","sequence":"additional","affiliation":[{"name":"China Telecom Cloud Technology Co. Ltd., Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1046-7997","authenticated-orcid":false,"given":"Hongqiang","family":"Xu","sequence":"additional","affiliation":[{"name":"China Telecom Cloud Technology Co. Ltd., Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8740-5180","authenticated-orcid":false,"given":"Yujun","family":"Wang","sequence":"additional","affiliation":[{"name":"China Telecom Cloud Technology Co. Ltd., Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2527-5049","authenticated-orcid":false,"given":"Liang","family":"Li","sequence":"additional","affiliation":[{"name":"China Telecom Cloud Computing Research Institute, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6436-9892","authenticated-orcid":false,"given":"Jiansong","family":"zhang","sequence":"additional","affiliation":[{"name":"China Telecom Cloud Computing Research Institute, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3472-1717","authenticated-orcid":false,"given":"Jie","family":"Wu","sequence":"additional","affiliation":[{"name":"China Telecom Cloud Computing Research Institute, Beijing, China and Temple University, Philadelphia, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,3,22]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2012.6168946"},{"key":"e_1_3_2_1_2_1","unstructured":"Advanced Micro Devices Inc. (AMD). 2024. ROCm Software Stack Documentation. https:\/\/rocm.docs.amd.com Accessed: 2024-06-05."},{"key":"e_1_3_2_1_3_1","volume-title":"Firecracker: Lightweight Virtualization for Serverless Applications. In 17th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2020","author":"Agache Alexandru","year":"2020","unstructured":"Alexandru Agache, Marc Brooker, Alexandra Iordache, Anthony Liguori, Rolf Neugebauer, Phil Piwonka, and Diana-Maria Popa. 2020. Firecracker: Lightweight Virtualization for Serverless Applications. In 17th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2020, Santa Clara, CA, USA, February 25-27, 2020, Ranjita Bhagwan and George Porter (Eds.). USENIX Association, 419-434. https:\/\/www.usenix.org\/conference\/nsdi20\/presentation\/agache"},{"key":"e_1_3_2_1_4_1","first-page":"923","volume-title":"Proceedings of the 2018 USENIX Annual Technical Conference, USENIX ATC 2018","author":"Akkus Istemi Ekin","year":"2018","unstructured":"Istemi Ekin Akkus, Ruichuan Chen, Ivica Rimac, Manuel Stein, Klaus Satzke, Andre Beck, Paarijaat Aditya, and Volker Hilt. 2018. SAND: Towards High-Performance Serverless Computing. In Proceedings of the 2018 USENIX Annual Technical Conference, USENIX ATC 2018, Boston, MA, USA, July 11-13, 2018, Haryadi S. Gunawi and Benjamin C. Reed (Eds.). USENIX Association, 923-935. https:\/\/www.usenix.org\/conference\/atc18\/presentation\/akkus"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00073"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3676641.3715988"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3492321.3524270"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00046"},{"key":"e_1_3_2_1_9_1","unstructured":"Microsoft Azure. 2025. Using serverless GPUs in Azure Container Apps. https:\/\/learn.microsoft.com\/en-us\/azure\/container-apps\/gpu-serverless-overview."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3623278.3624765"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3731569.3764843"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3037697.3037700"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607060"},{"key":"e_1_3_2_1_14_1","first-page":"199","volume-title":"Proceedings of the 2022 USENIX Annual Technical Conference, USENIX ATC 2022","author":"Choi Seungbeom","year":"2022","unstructured":"Seungbeom Choi, Sunho Lee, Yeonjae Kim, Jongse Park, Youngjin Kwon, and Jaehyuk Huh. 2022. Serving Heterogeneous Machine Learning Models on Multi-GPU Servers with Spatio-Temporal Sharing. In Proceedings of the 2022 USENIX Annual Technical Conference, USENIX ATC 2022, Carlsbad, CA, USA, July 11-13, 2022, Jiri Schindler and Noa Zilberman (Eds.). USENIX Association, 199-216. https:\/\/www.usenix.org\/conference\/atc22\/presentation\/choi-seungbeom"},{"key":"e_1_3_2_1_15_1","unstructured":"Alibaba Cloud. 2025a. Alibaba Function Compute (FC). https:\/\/www.aliyun.com\/product\/fc."},{"key":"e_1_3_2_1_16_1","unstructured":"Aliyun Cloud. 2025b. cGPU overview. https:\/\/www.alibabacloud.com\/help\/en\/container-service-for-kubernetes\/latest\/cgpu-overview."},{"key":"e_1_3_2_1_17_1","unstructured":"Alibaba Cloud. 2025c. Function Compute Service Pricing & Purchasing Methods. https:\/\/www.alibabacloud.com\/product\/function-compute\/pricing."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421285"},{"key":"e_1_3_2_1_19_1","volume-title":"Clipper: A Low-Latency Online Prediction Serving System. In 14th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2017","author":"Crankshaw Daniel","year":"2017","unstructured":"Daniel Crankshaw, Xin Wang, Giulio Zhou, Michael J. Franklin, Joseph E. Gonzalez, and Ion Stoica. 2017. Clipper: A Low-Latency Online Prediction Serving System. In 14th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2017, Boston, MA, USA, March 27-29, 2017, Aditya Akella and Jon Howell (Eds.). USENIX Association, 613-627. https:\/\/www.usenix.org\/conference\/nsdi17\/technical-sessions\/presentation\/crankshaw"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421284"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507732"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378512"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCS.2010.5547126"},{"key":"e_1_3_2_1_24_1","unstructured":"Hugging Face. 2025. The AI community building the future. https:\/\/huggingface.co\/models?sort=downloads."},{"key":"e_1_3_2_1_25_1","volume-title":"Fast and Slow: Low-Latency Video Processing Using Thousands of Tiny Threads. In 14th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2017","author":"Fouladi Sadjad","year":"2017","unstructured":"Sadjad Fouladi, Riad S. Wahby, Brennan Shacklett, Karthikeyan Balasubramaniam, William Zeng, Rahul Bhalerao, Anirudh Sivaraman, George Porter, and Keith Winstein. 2017. Encoding, Fast and Slow: Low-Latency Video Processing Using Thousands of Tiny Threads. In 14th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2017, Boston, MA, USA, March 27-29, 2017, Aditya Akella and Jon Howell (Eds.). USENIX Association, 363-376. https:\/\/www.usenix.org\/conference\/nsdi17\/technical-sessions\/presentation\/fouladi"},{"key":"e_1_3_2_1_26_1","volume-title":"ServerlessLLM: Low-Latency Serverless Inference for Large Language Models. In 18th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2024","author":"Fu Yao","year":"2024","unstructured":"Yao Fu, Leyang Xue, Yeqi Huang, Andrei-Octavian Brabete, Dmitrii Ustiugov, Yuvraj Patel, and Luo Mai. 2024. ServerlessLLM: Low-Latency Serverless Inference for Large Language Models. In 18th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2024, Santa Clara, CA, USA, July 10-12, 2024, Ada Gavrilovska and Douglas B. Terry (Eds.). USENIX Association, 135-153. https:\/\/www.usenix.org\/conference\/osdi24\/presentation\/fu"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3445814.3446757"},{"key":"e_1_3_2_1_28_1","unstructured":"Google. 2022. gVisor: Application Kernel for Containers. https:\/\/gvisor.dev\/."},{"key":"e_1_3_2_1_29_1","unstructured":"GoogleCloud. 2025. Serverless Optical Character Recognition (OCR) Tutorial. https:\/\/cloud.google.com\/functions\/docs\/tutorials\/ocr."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575721"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3605573.3605638"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3135974.3135993"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2024.3430063"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3629567"},{"key":"e_1_3_2_1_35_1","unstructured":"Docker Inc. 2025. docker container pause. https:\/\/docs.docker.com\/engine\/reference\/commandline\/container_pause\/."},{"key":"e_1_3_2_1_36_1","unstructured":"Andy Jassy. 2025. Amazon AWS ReInvent Keynote. https:\/\/www.youtube.com\/watch?v=ZOIkOnW640A."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3445814.3446701"},{"key":"e_1_3_2_1_38_1","unstructured":"Kaggle. 2025. Integrating TensorFlow Hub with Kaggle Models. https:\/\/www.kaggle.com\/models?tfhub-redirect=true&owner-type=organization."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3492321.3527539"},{"key":"e_1_3_2_1_40_1","first-page":"789","volume-title":"Proceedings of the 2018 USENIX Annual Technical Conference, USENIX ATC 2018","author":"Klimovic Ana","year":"2018","unstructured":"Ana Klimovic, Yawen Wang, Christos Kozyrakis, Patrick Stuedi, Jonas Pfefferle, and Animesh Trivedi. 2018. Understanding Ephemeral Storage for Serverless Analytics. In Proceedings of the 2018 USENIX Annual Technical Conference, USENIX ATC 2018, Boston, MA, USA, July 11-13, 2018, Haryadi S. Gunawi and Benjamin C. Reed (Eds.). USENIX Association, 789-794. https:\/\/www.usenix.org\/conference\/atc18\/presentation\/klimovic-serverless"},{"key":"e_1_3_2_1_41_1","first-page":"805","volume-title":"Proceedings of the 2021 USENIX Annual Technical Conference, USENIX ATC 2021","author":"Kotni Swaroop","year":"2021","unstructured":"Swaroop Kotni, Ajay Nayak, Vinod Ganapathy, and Arkaprava Basu. 2021. Faastlane: Accelerating Function-as-a-Service Workflows. In Proceedings of the 2021 USENIX Annual Technical Conference, USENIX ATC 2021, July 14-16, 2021, Irina Calciu and Geoff Kuenning (Eds.). USENIX Association, 805-820. https:\/\/www.usenix.org\/conference\/atc21\/presentation\/kotni"},{"key":"e_1_3_2_1_42_1","unstructured":"Kubernetes. 2025. Production-Grade Container Orchestration. https:\/\/kubernetes.io\/."},{"key":"e_1_3_2_1_43_1","unstructured":"AWS Lambda. 2025a. Keeping Functions Warm. https:\/\/docs.aws.amazon.com\/lambda\/latest\/dg\/lambda-concurrency.html."},{"key":"e_1_3_2_1_44_1","unstructured":"AWS Lambda. 2025b. Serverless Compute - Amazon Web Services. https:\/\/aws.amazon.com\/cn\/lambda\/pricing\/."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3342195.3387547"},{"key":"e_1_3_2_1_46_1","unstructured":"Kevin Lee Vijay Rao and William Arnold. 2025. Accelerating Facebook's Infrastructure with Application-Specific Hardware. https:\/\/engineering.fb.com\/2019\/03\/14\/data-center-engineering\/accelerating-infrastructure\/."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3542929.3563510"},{"key":"e_1_3_2_1_48_1","volume-title":"Proceedings of the 2022 USENIX Annual Technical Conference, USENIX ATC 2022","author":"Li Jie","year":"2022","unstructured":"Jie Li, Laiping Zhao, Yanan Yang, Kunlin Zhan, and Keqiu Li. 2022c. Tetris: Memory-efficient Serverless Inference through Tensor Sharing. In Proceedings of the 2022 USENIX Annual Technical Conference, USENIX ATC 2022, Carlsbad, CA, USA, July 11-13, 2022, Jiri Schindler and Noa Zilberman (Eds.). USENIX Association. https:\/\/www.usenix.org\/conference\/atc22\/presentation\/li-jie"},{"key":"e_1_3_2_1_49_1","volume-title":"MinFlow: High-performance and Cost-efficient Data Passing for I\/O-intensive Stateful Serverless Analytics. In 22nd USENIX Conference on File and Storage Technologies, FAST 2024","author":"Li Tao","year":"2024","unstructured":"Tao Li, Yongkun Li, Wenzhe Zhu, Yinlong Xu, and John C. S. Lui. 2024. MinFlow: High-performance and Cost-efficient Data Passing for I\/O-intensive Stateful Serverless Analytics. In 22nd USENIX Conference on File and Storage Technologies, FAST 2024, Santa Clara, CA, USA, February 27-29, 2024, Xiaosong Ma and Youjip Won (Eds.). USENIX Association, 311-327. https:\/\/www.usenix.org\/conference\/fast24\/presentation\/li"},{"key":"e_1_3_2_1_50_1","first-page":"69","volume-title":"Proceedings of the 2022 USENIX Annual Technical Conference, USENIX ATC 2022","author":"Li Zijun","year":"2022","unstructured":"Zijun Li, Linsong Guo, Quan Chen, Jiagan Cheng, Chuhao Xu, Deze Zeng, Zhuo Song, Tao Ma, Yong Yang, Chao Li, and Minyi Guo. 2022a. Help Rather Than Recycle: Alleviating Cold Startup in Serverless Computing Through Inter-Function Container Sharing. In Proceedings of the 2022 USENIX Annual Technical Conference, USENIX ATC 2022, Carlsbad, CA, USA, July 11-13, 2022, Jiri Schindler and Noa Zilberman (Eds.). USENIX Association, 69-84. https:\/\/www.usenix.org\/conference\/atc22\/presentation\/li-zijun-help"},{"key":"e_1_3_2_1_51_1","first-page":"161","volume-title":"Proceedings of the 2021 USENIX Annual Technical Conference, USENIX ATC 2021","author":"Lim Gangmuk","year":"2021","unstructured":"Gangmuk Lim, Jeongseob Ahn, Wencong Xiao, Youngjin Kwon, and Myeongjae Jeon. 2021. Zico: Efficient GPU Memory Sharing for Concurrent DNN Training. In Proceedings of the 2021 USENIX Annual Technical Conference, USENIX ATC 2021, July 14-16, 2021, Irina Calciu and Geoff Kuenning (Eds.). USENIX Association, 161-175. https:\/\/www.usenix.org\/conference\/atc21\/presentation\/lim"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3669940.3707251"},{"key":"e_1_3_2_1_53_1","unstructured":"LWN.net. 2025. vfio\/mdev: IOMMU aware mediated device. https:\/\/lwn.net\/Articles\/783892\/. (2025)."},{"key":"e_1_3_2_1_54_1","volume-title":"Themis: Fair and Efficient GPU Cluster Scheduling. In 17th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2020","author":"Mahajan Kshiteej","year":"2020","unstructured":"Kshiteej Mahajan, Arjun Balasubramanian, Arjun Singhvi, Shivaram Venkataraman, Aditya Akella, Amar Phanishayee, and Shuchi Chawla. 2020. Themis: Fair and Efficient GPU Cluster Scheduling. In 17th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2020, Santa Clara, CA, USA, February 25-27, 2020, Ranjita Bhagwan and George Porter (Eds.). USENIX Association, 289-304. https:\/\/www.usenix.org\/conference\/nsdi20\/presentation\/mahajan"},{"key":"e_1_3_2_1_55_1","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2022","author":"Mahgoub Ashraf","year":"2022","unstructured":"Ashraf Mahgoub, Edgardo Barsallo Yi, Karthick Shankar, Sameh Elnikety, Somali Chaterji, and Saurabh Bagchi. 2022. ORION and the Three Rights: Sizing, Bundling, and Prewarming for Serverless DAGs. In 16th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2022, Carlsbad, CA, USA, July 11-13, 2022, Marcos K. Aguilera and Hakim Weatherspoon (Eds.). USENIX Association, 303-320. https:\/\/www.usenix.org\/conference\/osdi22\/presentation\/mahgoub"},{"key":"e_1_3_2_1_56_1","unstructured":"Philipp Muens. 2025. Serverless Facebook Messenger Bot. https:\/\/github.com\/pmuens\/serverless-facebook-messenger-bot."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3318464.3389758"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613163"},{"key":"e_1_3_2_1_59_1","unstructured":"NVIDIA. 2025a. CUDA checkpoint and restore utility. https:\/\/github.com\/NVIDIA\/cuda-checkpoint."},{"key":"e_1_3_2_1_60_1","unstructured":"NVIDIA. 2025b. Multi-Instance GPU (MIG). https:\/\/www.nvidia.com\/en-us\/technologies\/multi-instance-gpu\/."},{"key":"e_1_3_2_1_61_1","unstructured":"NVIDIA. 2025c. Multi-Process Service (MPS). https:\/\/docs.nvidia.com\/deploy\/mps\/index.html."},{"key":"e_1_3_2_1_62_1","unstructured":"NVIDIA. 2025d. NVIDIA Ampere GPU Architecture Tuning Guide. https:\/\/docs.nvidia.com\/cuda\/archive\/11.0\/ampere-tuning-guide\/index.html."},{"key":"e_1_3_2_1_63_1","unstructured":"NVIDIA. 2025 e. Scheduling policy of NVIDIA's vGPU. https:\/\/docs.nvidia.com\/vgpu\/faq\/latest\/deployment.html."},{"key":"e_1_3_2_1_64_1","unstructured":"NVIDIA. 2025 f. Time-Slicing GPUs in Kubernetes. https:\/\/docs.nvidia.com\/datacenter\/cloud-native\/gpu-operator\/latest\/gpu-sharing.html."},{"key":"e_1_3_2_1_65_1","unstructured":"NVIDIA. 2025 g. Using NVIDIA vGPU. https:\/\/docs.nvidia.com\/datacenter\/cloud-native\/gpu-operator\/latest\/install-gpu-operator-vgpu.html."},{"key":"e_1_3_2_1_66_1","volume-title":"Proceedings of the 2018 USENIX Annual Technical Conference, USENIX ATC 2018","author":"Oakes Edward","year":"2018","unstructured":"Edward Oakes, Leon Yang, Dennis Zhou, Kevin Houck, Tyler Harter, Andrea C. Arpaci-Dusseau, and Remzi H. Arpaci-Dusseau. 2018. SOCK: Rapid Task Provisioning with Serverless-Optimized Containers. In Proceedings of the 2018 USENIX Annual Technical Conference, USENIX ATC 2018, Boston, MA, USA, July 11-13, 2018, Haryadi S. Gunawi and Benjamin C. Reed (Eds.). USENIX Association, 57-70. https:\/\/www.usenix.org\/conference\/atc18\/presentation\/oakes"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620678.3624664"},{"key":"e_1_3_2_1_68_1","volume-title":"Fast and Slow: Scalable Analytics on Serverless Infrastructure. In 16th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2019","author":"Pu Qifan","year":"2019","unstructured":"Qifan Pu, Shivaram Venkataraman, and Ion Stoica. 2019. Shuffling, Fast and Slow: Scalable Analytics on Serverless Infrastructure. In 16th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2019, Boston, MA, February 26-28, 2019, Jay R. Lorch and Minlan Yu (Eds.). USENIX Association, 193-206. https:\/\/www.usenix.org\/conference\/nsdi19\/presentation\/pu"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/IOTSMS48152.2019.8939164"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA45697.2020.00045"},{"key":"e_1_3_2_1_71_1","first-page":"397","volume-title":"Proceedings of the 2021 USENIX Annual Technical Conference, USENIX ATC 2021","author":"Romero Francisco","year":"2021","unstructured":"Francisco Romero, Qian Li, Neeraja J. Yadwadkar, and Christos Kozyrakis. 2021. INFaaS: Automated Model-less Inference Serving. In Proceedings of the 2021 USENIX Annual Technical Conference, USENIX ATC 2021, July 14-16, 2021, Irina Calciu and Geoff Kuenning (Eds.). USENIX Association, 397-411. https:\/\/www.usenix.org\/conference\/atc21\/presentation\/romero"},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1145\/585265.585269"},{"key":"e_1_3_2_1_73_1","unstructured":"Amazon Web Services. 2025a. Amazon EC2 T3 Instances. https:\/\/aws.amazon.com\/ec2\/instance-types\/t3\/."},{"key":"e_1_3_2_1_74_1","unstructured":"Amazon Web Services. 2025b. Amazon GPU EC2 Instance Pricing. https:\/\/calculator.aws\/#\/createCalculator\/ec2-enhancement."},{"key":"e_1_3_2_1_75_1","unstructured":"Amazon Web Services. 2025c. AWS Lambda overview. https:\/\/aws.amazon.com\/lambda\/."},{"key":"e_1_3_2_1_76_1","unstructured":"Amazon Web Sevice. 2025. Serverless Application Lens: Alexa Skills. https:\/\/docs.aws.amazon.com\/wellarchitected\/latest\/serverless-applications-lens\/alexa-skills.html."},{"key":"e_1_3_2_1_77_1","first-page":"205","volume-title":"Proceedings of the 2020 USENIX Annual Technical Conference, USENIX ATC 2020","author":"Shahrad Mohammad","year":"2020","unstructured":"Mohammad Shahrad, Rodrigo Fonseca, I n, igo Goiri, Gohar Irfan Chaudhry, Paul Batum, Jason Cooke, Eduardo Laureano, Colby Tresness, Mark Russinovich, and Ricardo Bianchini. 2020. Serverless in the Wild: Characterizing and Optimizing the Serverless Workload at a Large Cloud Provider. In Proceedings of the 2020 USENIX Annual Technical Conference, USENIX ATC 2020, July 15-17, 2020, Ada Gavrilovska and Erez Zadok (Eds.). USENIX Association, 205-218. https:\/\/www.usenix.org\/conference\/atc20\/presentation\/shahrad"},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359658"},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2011.112"},{"key":"e_1_3_2_1_80_1","volume-title":"Hellerstein","author":"Sreekanti Vikram","year":"2020","unstructured":"Vikram Sreekanti, Harikaran Subbaraj, Chenggang Wu, Joseph E. Gonzalez, and Joseph M. Hellerstein. 2020. Optimizing Prediction Serving on Low-Latency Serverless Dataflow. CoRR, Vol. abs\/2007.05832 (2020). arXiv:2007.05832 https:\/\/arxiv.org\/abs\/2007.05832"},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589069"},{"key":"e_1_3_2_1_82_1","unstructured":"Radostin Stoyanov Vikt\u00f3ria Spi\u0161akov\u00e1 Jesus Ramos Steven Gurfinkel Andrei Vagin Adrian Reber Wesley Armour and Rodrigo Bruno. 2025. CRIUgpu: Transparent Checkpointing of GPU-Accelerated Workloads. arXiv:2502.16631 [cs.DC] https:\/\/arxiv.org\/abs\/2502.16631"},{"key":"e_1_3_2_1_83_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3629578"},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"publisher","DOI":"10.1145\/3698038.3698509"},{"key":"e_1_3_2_1_85_1","doi-asserted-by":"publisher","DOI":"10.1145\/3673038.3673138"},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"publisher","DOI":"10.1145\/3542929.3563476"},{"key":"e_1_3_2_1_87_1","unstructured":"Slurm Development Team. 2025. Slurm Quickstart Guide. https:\/\/slurm.schedmd.com\/quickstart.html."},{"key":"e_1_3_2_1_88_1","first-page":"495","volume-title":"15th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2021","author":"Thorpe John","year":"2021","unstructured":"John Thorpe, Yifan Qiao, Jonathan Eyolfson, Shen Teng, Guanzhou Hu, Zhihao Jia, Jinliang Wei, Keval Vora, Ravi Netravali, Miryung Kim, and Guoqing Harry Xu. 2021. Dorylus: Affordable, Scalable, and Accurate GNN Training with Distributed CPU Servers and Serverless Threads. In 15th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2021, July 14-16, 2021, Angela Demke Brown and Jay R. Lorch (Eds.). USENIX Association, 495-514. https:\/\/www.usenix.org\/conference\/osdi21\/presentation\/thorpe"},{"key":"e_1_3_2_1_89_1","doi-asserted-by":"publisher","DOI":"10.1145\/3445814.3446714"},{"key":"e_1_3_2_1_90_1","volume-title":"18th USENIX Conference on File and Storage Technologies, FAST 2020","author":"Wang Ao","year":"2020","unstructured":"Ao Wang, Jingyuan Zhang, Xiaolong Ma, Ali Anwar, Lukas Rupprecht, Dimitrios Skourtis, Vasily Tarasov, Feng Yan, and Yue Cheng. 2020. InfiniCache: Exploiting Ephemeral Serverless Functions to Build a Cost-Effective Memory Cache. In 18th USENIX Conference on File and Storage Technologies, FAST 2020, Santa Clara, CA, USA, February 24-27, 2020, Sam H. Noh and Brent Welch (Eds.). USENIX Association, 267-281. https:\/\/www.usenix.org\/conference\/fast20\/presentation\/wang-ao"},{"key":"e_1_3_2_1_91_1","doi-asserted-by":"publisher","DOI":"10.1145\/3431379.3460646"},{"key":"e_1_3_2_1_92_1","volume-title":"Proceedings of the Fourth Conference on Machine Learning and Systems, MLSys 2021","author":"Wang Guanhua","year":"2021","unstructured":"Guanhua Wang, Kehan Wang, Kenan Jiang, Xiangjun Li, and Ion Stoica. 2021b. Wavelet: Efficient DNN Training with Tick-Tock Scheduling. In Proceedings of the Fourth Conference on Machine Learning and Systems, MLSys 2021, virtual, April 5-9, 2021, Alex Smola, Alex Dimakis, and Ion Stoica (Eds.). mlsys.org. https:\/\/proceedings.mlsys.org\/paper_files\/paper\/2021\/hash\/099268c3121d49937a67a052c51f865d-Abstract.html"},{"key":"e_1_3_2_1_93_1","volume-title":"No Provisioned Concurrency: Fast RDMA-codesigned Remote Fork for Serverless Computing. In 17th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2023","author":"Wei Xingda","year":"2023","unstructured":"Xingda Wei, Fangming Lu, Tianxia Wang, Jinyu Gu, Yuhan Yang, Rong Chen, and Haibo Chen. 2023. No Provisioned Concurrency: Fast RDMA-codesigned Remote Fork for Serverless Computing. In 17th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2023, Boston, MA, USA, July 10-12, 2023, Roxana Geambasu and Ed Nightingale (Eds.). USENIX Association, 497-517. https:\/\/www.usenix.org\/conference\/osdi23\/presentation\/wei-rdma"},{"key":"e_1_3_2_1_94_1","volume-title":"Transparent GPU Sharing in Container Clouds for Deep Learning Workloads. In 20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023","author":"Wu Bingyang","year":"2023","unstructured":"Bingyang Wu, Zili Zhang, Zhihao Bai, Xuanzhe Liu, and Xin Jin. 2023. Transparent GPU Sharing in Container Clouds for Deep Learning Workloads. In 20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023, Boston, MA, April 17-19, 2023, Mahesh Balakrishnan and Manya Ghobadi (Eds.). USENIX Association, 69-85. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/wu"},{"key":"e_1_3_2_1_95_1","doi-asserted-by":"publisher","DOI":"10.1145\/3514221.3517905"},{"key":"e_1_3_2_1_96_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3650074"},{"key":"e_1_3_2_1_97_1","doi-asserted-by":"publisher","DOI":"10.1145\/3698038.3698510"},{"key":"e_1_3_2_1_98_1","doi-asserted-by":"publisher","DOI":"10.1145\/3623278.3624769"},{"key":"e_1_3_2_1_99_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507709"},{"key":"e_1_3_2_1_100_1","doi-asserted-by":"publisher","DOI":"10.1145\/3018743.3018754"},{"key":"e_1_3_2_1_101_1","doi-asserted-by":"publisher","DOI":"10.1145\/3617232.3624871"},{"key":"e_1_3_2_1_102_1","volume-title":"Not the Function: Rethinking Function Orchestration in Serverless Computing. In 20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023","author":"Yu Minchen","year":"2023","unstructured":"Minchen Yu, Tingjia Cao, Wei Wang, and Ruichuan Chen. 2023a. Following the Data, Not the Function: Rethinking Function Orchestration in Serverless Computing. In 20th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2023, Boston, MA, April 17-19, 2023, Mahesh Balakrishnan and Manya Ghobadi (Eds.). USENIX Association, 1489-1504. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/yu"},{"key":"e_1_3_2_1_103_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2306.03622"},{"key":"e_1_3_2_1_104_1","first-page":"597","volume-title":"Resource-Efficient Inference. In Proceedings of the 2025 USENIX Annual Technical Conference, USENIX ATC 2025","author":"Yu Minchen","year":"2025","unstructured":"Minchen Yu, Ao Wang, Dong Chen, Haoxuan Yu, Xiaonan Luo, Zhuohao Li, Wei Wang, Ruichuan Chen, Dapeng Nie, Haoran Yang, and Yu Ding. 2025. Torpor: GPU-Enabled Serverless Computing for Low-Latency, Resource-Efficient Inference. In Proceedings of the 2025 USENIX Annual Technical Conference, USENIX ATC 2025, Boston, MA, USA, July 7-9, 2025, Deniz Altinb\u00fcken and Ryan Stutsman (Eds.). USENIX Association, 597-612. https:\/\/www.usenix.org\/conference\/atc25\/presentation\/yu"},{"key":"e_1_3_2_1_105_1","first-page":"1049","volume-title":"Proceedings of the 2019 USENIX Annual Technical Conference, USENIX ATC 2019","author":"Zhang Chengliang","year":"2019","unstructured":"Chengliang Zhang, Minchen Yu, Wei Wang, and Feng Yan. 2019. MArk: Exploiting Cloud Services for Cost-Effective, SLO-Aware Machine Learning Inference Serving. In Proceedings of the 2019 USENIX Annual Technical Conference, USENIX ATC 2019, Renton, WA, USA, July 10-12, 2019, Dahlia Malkhi and Dan Tsafrir (Eds.). USENIX Association, 1049-1062. https:\/\/www.usenix.org\/conference\/atc19\/presentation\/zhang-chengliang"},{"key":"e_1_3_2_1_106_1","first-page":"951","volume-title":"Proceedings of the 2018 USENIX Annual Technical Conference, USENIX ATC 2018","author":"Zhang Minjia","year":"2018","unstructured":"Minjia Zhang, Samyam Rajbhandari, Wenhan Wang, and Yuxiong He. 2018. DeepCPU: Serving RNN-based Deep Learning Models 10x Faster. In Proceedings of the 2018 USENIX Annual Technical Conference, USENIX ATC 2018, Boston, MA, USA, July 11-13, 2018, Haryadi S. Gunawi and Benjamin C. Reed (Eds.). USENIX Association, 951-965. https:\/\/www.usenix.org\/conference\/atc18\/presentation\/zhang-minjia"},{"key":"e_1_3_2_1_107_1","doi-asserted-by":"publisher","DOI":"10.1145\/3342195.3387534"},{"key":"e_1_3_2_1_108_1","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575752"},{"key":"e_1_3_2_1_109_1","doi-asserted-by":"publisher","DOI":"10.1145\/3567955.3567960"}],"event":{"name":"ASPLOS '26: 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems","location":"Pittsburgh PA USA","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems","SIGPLAN ACM Special Interest Group on Programming Languages","SIGARCH ACM Special Interest Group on Computer Architecture","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2"],"original-title":[],"deposited":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T14:07:44Z","timestamp":1773583664000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3779212.3790168"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,22]]},"references-count":109,"alternative-id":["10.1145\/3779212.3790168","10.1145\/3779212"],"URL":"https:\/\/doi.org\/10.1145\/3779212.3790168","relation":{},"subject":[],"published":{"date-parts":[[2026,3,22]]},"assertion":[{"value":"2026-03-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}