{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T01:12:23Z","timestamp":1780708343128,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":79,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,13]]},"DOI":"10.1145\/3731569.3764813","type":"proceedings-article","created":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T12:43:24Z","timestamp":1759322604000},"page":"996-1013","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["PhoenixOS: Concurrent OS-level GPU Checkpoint and Restore with Validated Speculation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4983-6047","authenticated-orcid":false,"given":"Xingda","family":"Wei","sequence":"first","affiliation":[{"name":"Institute of Parallel and Distributed Systems, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2251-4146","authenticated-orcid":false,"given":"Zhuobin","family":"Huang","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-0083-6964","authenticated-orcid":false,"given":"Tianle","family":"Sun","sequence":"additional","affiliation":[{"name":"Institute of Parallel and Distributed Systems, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8042-0863","authenticated-orcid":false,"given":"Yingyi","family":"Hao","sequence":"additional","affiliation":[{"name":"Institute of Parallel and Distributed Systems, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6115-8130","authenticated-orcid":false,"given":"Rong","family":"Chen","sequence":"additional","affiliation":[{"name":"Institute of Parallel and Distributed Systems, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1536-7485","authenticated-orcid":false,"given":"Mingcong","family":"Han","sequence":"additional","affiliation":[{"name":"Institute of Parallel and Distributed Systems, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8112-8481","authenticated-orcid":false,"given":"Jinyu","family":"Gu","sequence":"additional","affiliation":[{"name":"Institute of Parallel and Distributed Systems, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9720-0361","authenticated-orcid":false,"given":"Haibo","family":"Chen","sequence":"additional","affiliation":[{"name":"Institute of Parallel and Distributed Systems, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,12]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"https:\/\/docs.nvidia.com\/cuda\/cuda-binary-utilities\/index.html#cu-filt","year":"2025","unstructured":"cu++filt. https:\/\/docs.nvidia.com\/cuda\/cuda-binary-utilities\/index.html#cu-filt, 2025."},{"key":"e_1_3_2_1_2_1","volume-title":"https:\/\/docs.nvidia.com\/cuda\/gpudirect-rdma\/","author":"Developing","year":"2025","unstructured":"Developing a linux kernel module using gpudirect rdma. https:\/\/docs.nvidia.com\/cuda\/gpudirect-rdma\/, 2025."},{"key":"e_1_3_2_1_3_1","volume-title":"Kernel Library for LLM Serving. https:\/\/github.com\/flashinfer-ai\/flashinfer","author":"FlashInfer","year":"2025","unstructured":"FlashInfer: Kernel Library for LLM Serving. https:\/\/github.com\/flashinfer-ai\/flashinfer, 2025."},{"key":"e_1_3_2_1_4_1","volume-title":"https:\/\/developer.nvidia.com\/blog\/cuda-graphs\/","author":"Getting","year":"2025","unstructured":"Getting started with cuda graphs. https:\/\/developer.nvidia.com\/blog\/cuda-graphs\/, 2025."},{"key":"e_1_3_2_1_5_1","first-page":"746","volume-title":"EuroSys '22: Seventeenth European Conference on Computer Systems","author":"Ao L.","year":"2022","unstructured":"Ao, L., Porter, G., and Voelker, G. M. Faasnap: Faas made fast using snapshot-based vms. In EuroSys '22: Seventeenth European Conference on Computer Systems, Rennes, France, April 5-8, 2022 (2022), Y. Bromberg, A. Kermarrec, and C. Kozyrakis, Eds., ACM, pp. 730\u2013746."},{"key":"e_1_3_2_1_6_1","first-page":"514","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20)","author":"Bai Z.","year":"2020","unstructured":"Bai, Z., Zhang, Z., Zhu, Y., and Jin, X. PipeSwitch: Fast pipelined context switching for deep learning applications. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20) (Nov. 2020), USENIX Association, pp. 499\u2013514."},{"key":"e_1_3_2_1_7_1","unstructured":"Bakita J. and Anderson J. H. Demystifying nvidia gpu internals to enable reliable gpu management."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1145\/2743017","article-title":"The design and implementation of a verification technique for GPU kernels","volume":"37","author":"Betts A.","year":"2015","unstructured":"Betts, A., Chong, N., Donaldson, A. F., Ketema, J., Qadeer, S., Thomson, P., and Wickerson, J. The design and implementation of a verification technique for GPU kernels. ACM Trans. Program. Lang. Syst. 37, 3 (2015), 10:1\u201310:49.","journal-title":"ACM Trans. Program. Lang. Syst."},{"key":"e_1_3_2_1_9_1","first-page":"132","volume-title":"Proceedings of the 27th Annual ACM SIGPLAN Conference on Object-Oriented Programming, Systems, Languages, and Applications, OOPSLA 2012, part of SPLASH 2012","author":"Betts A.","year":"2012","unstructured":"Betts, A., Chong, N., Donaldson, A. F., Qadeer, S., and Thomson, P. Gpuverify: a verifier for GPU kernels. In Proceedings of the 27th Annual ACM SIGPLAN Conference on Object-Oriented Programming, Systems, Languages, and Applications, OOPSLA 2012, part of SPLASH 2012, Tucson, AZ, USA, October 21-25, 2012 (2012), G. T. Leavens and M. B. Dwyer, Eds., ACM, pp. 113\u2013132."},{"key":"e_1_3_2_1_10_1","first-page":"54","volume-title":"Proceedings of the 2009 IEEE International Symposium on Workload Characterization, IISWC 2009","author":"Che S.","year":"2009","unstructured":"Che, S., Boyer, M., Meng, J., Tarjan, D., Sheaffer, J. W., Lee, S., and Skadron, K. Rodinia: A benchmark suite for heterogeneous computing. In Proceedings of the 2009 IEEE International Symposium on Workload Characterization, IISWC 2009, October 4-6, 2009, Austin, TX, USA (2009), IEEE Computer Society, pp. 44\u201354."},{"key":"e_1_3_2_1_11_1","first-page":"594","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2018","author":"Chen T.","year":"2018","unstructured":"Chen, T., Moreau, T., Jiang, Z., Zheng, L., Yan, E. Q., Shen, H., Cowan, M., Wang, L., Hu, Y., Ceze, L., Guestrin, C., and Krishnamurthy, A. TVM: an automated end-to-end optimizing compiler for deep learning. In 13th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2018, Carlsbad, CA, USA, October 8-10, 2018 (2018), A. C. Arpaci-Dusseau and G. Voelker, Eds., USENIX Association, pp. 578\u2013594."},{"key":"e_1_3_2_1_12_1","volume-title":"5th International Symposium, NFM 2013, Moffett Field,, CA, USA, May 14-16, 2013. Proceedings","volume":"7871","author":"Chiang W.","year":"2013","unstructured":"Chiang, W., Gopalakrishnan, G., Li, G., and Rakamaric, Z. Formal analysis of GPU programs with atomics via conflict-directed delay-bounding. In NASA Formal Methods, 5th International Symposium, NFM 2013, Moffett Field,, CA, USA, May 14-16, 2013. Proceedings (2013), G. Brat, N. Rungta, and A. Venet, Eds., vol. 7871 of Lecture Notes in Computer Science, Springer, pp. 213\u2013228."},{"key":"e_1_3_2_1_13_1","volume-title":"Clang: a c language family frontend for llvm","year":"2024","unstructured":"clang. Clang: a c language family frontend for llvm, 2024."},{"key":"e_1_3_2_1_14_1","volume-title":"2nd Symposium on Networked Systems Design and Implementation NSDI (2005)","author":"Clark C.","year":"2005","unstructured":"Clark, C., Fraser, K., Hand, S., Hansen, J. G., Jul, E., Limpach, C., Pratt, I., and Warfield, A. Live migration of virtual machines. In 2nd Symposium on Networked Systems Design and Implementation NSDI (2005), May 2-4, 2005, Boston, Massachusetts, USA, Proceedings (2005), A. Vahdat and D. Wetherall, Eds., USENIX."},{"key":"e_1_3_2_1_15_1","volume-title":"https:\/\/developer.aliyun.com\/article\/1610603","author":"Cloud A.","year":"2024","unstructured":"Cloud, A. https:\/\/developer.aliyun.com\/article\/1610603, 2024."},{"key":"e_1_3_2_1_16_1","volume-title":"Serverless gpu overview. https:\/\/www.alibabacloud.com\/tech-news\/a\/serverless\/4o2cc4hux4q-serverless-gpu-overview","year":"2024","unstructured":"cloud, A. Serverless gpu overview. https:\/\/www.alibabacloud.com\/tech-news\/a\/serverless\/4o2cc4hux4q-serverless-gpu-overview, 2024."},{"key":"e_1_3_2_1_17_1","unstructured":"CRIU. Tcp repair mode in kernel."},{"key":"e_1_3_2_1_18_1","unstructured":"CRIU. Criu main page 2024."},{"key":"e_1_3_2_1_19_1","volume-title":"https:\/\/criu.org\/Memory_changes_tracking","author":"Memory","year":"2025","unstructured":"CRIU. Memory changes tracking. https:\/\/criu.org\/Memory_changes_tracking, 2025."},{"key":"e_1_3_2_1_20_1","first-page":"481","volume-title":"ASPLOS '20: Architectural Support for Programming Languages and Operating Systems","author":"Du D.","year":"2020","unstructured":"Du, D., Yu, T., Xia, Y., Zang, B., Yan, G., Qin, C., Wu, Q., and Chen, H. Catalyzer: Sub-millisecond startup for serverless computing with initialization-less booting. In ASPLOS '20: Architectural Support for Programming Languages and Operating Systems, Lausanne, Switzerland, March 16-20, 2020 (2020), J. R. Larus, L. Ceze, and K. Strauss, Eds., ACM, pp. 467\u2013481."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1007\/s11227-013-0884-0","article-title":"A survey of fault tolerance mechanisms and checkpoint\/restart implementations for high performance computing systems","volume":"65","author":"Egwutuoha I. P.","year":"2013","unstructured":"Egwutuoha, I. P., Levy, D., Selic, B., and Chen, S. A survey of fault tolerance mechanisms and checkpoint\/restart implementations for high performance computing systems. The Journal of Supercomputing 65, 3 (2013), 1302\u20131326.","journal-title":"The Journal of Supercomputing"},{"key":"e_1_3_2_1_22_1","first-page":"943","volume-title":"19th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2022","author":"Eisenman A.","year":"2022","unstructured":"Eisenman, A., Matam, K. K., Ingram, S., Mudigere, D., Krishnamoorthi, R., Nair, K., Smelyanskiy, M., and Annavaram, M. Check-n-run: a checkpointing system for training deep learning recommendation models. In 19th USENIX Symposium on Networked Systems Design and Implementation, NSDI 2022, Renton, WA, USA, April 4-6, 2022 (2022), A. Phanishayee and V. Sekar, Eds., USENIX Association, pp. 929\u2013943."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1145\/568522.568525","article-title":"A survey of rollback-recovery protocols in message-passing systems","volume":"34","author":"Elnozahy E. N.","year":"2002","unstructured":"Elnozahy, E. N., Alvisi, L., Wang, Y., and Johnson, D. B. A survey of rollback-recovery protocols in message-passing systems. ACM Comput. Surv. 34, 3 (2002), 375\u2013408.","journal-title":"ACM Comput. Surv."},{"key":"e_1_3_2_1_24_1","volume-title":"The ai community building the future. https:\/\/huggingface.co","author":"Face H.","year":"2024","unstructured":"Face, H. The ai community building the future. https:\/\/huggingface.co, 2024."},{"key":"e_1_3_2_1_25_1","unstructured":"FasterTransformer. Nvidia 2024."},{"key":"e_1_3_2_1_26_1","first-page":"466","volume-title":"45th IEEE\/ACM International Conference on Software Engineering: Software Engineering in Practice, SEIP@ICSE 2023","author":"Gao Y.","year":"2023","unstructured":"Gao, Y., Shi, X., Lin, H., Zhang, H., Wu, H., Li, R., and Yang, M. An empirical study on quality issues of deep learning platform. In 45th IEEE\/ACM International Conference on Software Engineering: Software Engineering in Practice, SEIP@ICSE 2023, Melbourne, Australia, May 14-20, 2023 (2023), IEEE, pp. 455\u2013466."},{"key":"e_1_3_2_1_27_1","volume-title":"33rd USENIX Security Symposium, USENIX Security 2024","author":"Guo Y.","year":"2024","unstructured":"Guo, Y., Zhang, Z., and Yang, J. GPU memory exploitation for fun and profit. In 33rd USENIX Security Symposium, USENIX Security 2024, Philadelphia, PA, USA, August 14-16, 2024 (2024), D. Balzarotti and W. Xu, Eds., USENIX Association."},{"key":"e_1_3_2_1_28_1","first-page":"1125","volume-title":"Proceedings of the Nineteenth European Conference on Computer Systems, EuroSys 2024","author":"Gupta T.","year":"2024","unstructured":"Gupta, T., Krishnan, S., Kumar, R., Vijeev, A., Gulavani, B. S., Kwatra, N., Ramjee, R., and Sivathanu, M. Just-in-time checkpointing: Low cost error recovery from deep learning training failures. In Proceedings of the Nineteenth European Conference on Computer Systems, EuroSys 2024, Athens, Greece, April 22-25, 2024 (2024), ACM, pp. 1110\u20131125."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"crossref","first-page":"8","DOI":"10.1145\/858336.858337","article-title":"Keykos architecture","volume":"19","author":"Hardy N","unstructured":"Hardy, N. Keykos architecture. SIGOPS Oper. Syst. Rev. 19, 4 (oct 1985), 8\u201325.","journal-title":"SIGOPS Oper. Syst. Rev."},{"key":"e_1_3_2_1_30_1","first-page":"067","volume-title":"Journal of Physics: Conference Series","volume":"46","author":"Hargrove P. H.","year":"2006","unstructured":"Hargrove, P. H., and Duell, J. C. Berkeley lab checkpoint\/restart (blcr) for linux clusters. In Journal of Physics: Conference Series (2006), vol. 46, IOP Publishing, p. 067."},{"key":"e_1_3_2_1_31_1","volume-title":"PARALLELGPUOS: A concurrent os-level GPU checkpoint and restore system using validated speculation. CoRR abs\/2405.12079v1","author":"Huang Z.","year":"2024","unstructured":"Huang, Z., Wei, X., Hao, Y., Chen, R., Han, M., Gu, J., and Chen, H. PARALLELGPUOS: A concurrent os-level GPU checkpoint and restore system using validated speculation. CoRR abs\/2405.12079v1 (2024)."},{"key":"e_1_3_2_1_32_1","volume-title":"21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24)","author":"Jiang Z.","year":"2024","unstructured":"Jiang, Z., Lin, H., Zhong, Y., Huang, Q., Chen, Y., Zhang, Z., Peng, Y., Li, X., Xie, C., Nong, S., Jia, Y., He, S., Chen, H., Bai, Z., Hou, Q., Yan, S., Zhou, D., Sheng, Y., Jiang, Z., Xu, H., Wei, H., Zhang, Z., Nie, P., Zou, L., Zhao, S., Xiang, L., Liu, Z., Li, Z., Jia, X., Ye, J., Jin, X., and Liu, X. MegaScale: Scaling large language model training to more than 10,000 GPUs. In 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24) (Santa Clara, CA, Apr. 2024), USENIX Association, pp. 745\u2013760."},{"key":"e_1_3_2_1_33_1","volume-title":"Starting up faster with aws lambda snap-start. https:\/\/aws.amazon.com\/cn\/blogs\/compute\/starting-up-faster-with-aws-lambda-snapstart\/","author":"Johnson E.","year":"2024","unstructured":"Johnson, E. Starting up faster with aws lambda snap-start. https:\/\/aws.amazon.com\/cn\/blogs\/compute\/starting-up-faster-with-aws-lambda-snapstart\/, 2024."},{"key":"e_1_3_2_1_34_1","volume-title":"Cloud programming simplified: A berkeley view on serverless computing. CoRR abs\/1902.03383","author":"Jonas E.","year":"2019","unstructured":"Jonas, E., Schleier-Smith, J., Sreekanti, V., Tsai, C., Khandelwal, A., Pu, Q., Shankar, V., Carreira, J., Krauth, K., Yadwadkar, N. J., Gonzalez, J. E., Popa, R. A., Stoica, I., and Patterson, D. A. Cloud programming simplified: A berkeley view on serverless computing. CoRR abs\/1902.03383 (2019)."},{"key":"e_1_3_2_1_35_1","first-page":"65","volume-title":"SOSP '21: ACM SIGOPS 28th Symposium on Operating Systems Principles, Virtual Event \/ Koblenz, Germany","author":"Kamath A. K.","year":"2021","unstructured":"Kamath, A. K., and Basu, A. iguard: In-gpu advanced race detection. In SOSP '21: ACM SIGOPS 28th Symposium on Operating Systems Principles, Virtual Event \/ Koblenz, Germany, October 26-29, 2021 (2021), R. van Renesse and N. Zeldovich, Eds., ACM, pp. 49\u201365."},{"key":"e_1_3_2_1_36_1","first-page":"626","volume-title":"Proceedings of the 29th Symposium on Operating Systems Principles, SOSP 2023","author":"Kwon W.","year":"2023","unstructured":"Kwon, W., Li, Z., Zhuang, S., Sheng, Y., Zheng, L., Yu, C. H., Gonzalez, J., Zhang, H., and Stoica, I. Efficient memory management for large language model serving with pagedattention. In Proceedings of the 29th Symposium on Operating Systems Principles, SOSP 2023, Koblenz, Germany, October 23-26, 2023 (2023), J. Flinn, M. I. Seltzer, P. Druschel, A. Kaufmann, and J. Mace, Eds., ACM, pp. 611\u2013626."},{"key":"e_1_3_2_1_37_1","volume-title":"Linux Symposium","volume":"159","author":"Laadan O.","year":"2010","unstructured":"Laadan, O., and Hallyn, S. E. Linux-cr: Transparent application checkpoint-restart in linux. In Linux Symposium (2010), vol. 159, Citeseer."},{"key":"e_1_3_2_1_38_1","first-page":"183","volume-title":"Proceedings of the ACM International Conference on Supercomputing, ICS 2019","author":"Lee K.","year":"2019","unstructured":"Lee, K., Sullivan, M. B., Hari, S. K. S., Tsai, T., Keckler, S. W., and Erez, M. GPU snapshot: checkpoint offloading for gpu-dense systems. In Proceedings of the ACM International Conference on Supercomputing, ICS 2019, Phoenix, AZ, USA, June 26-28, 2019 (2019), R. Eigenmann, C. Ding, and S. A. McKee, Eds., ACM, pp. 171\u2013183."},{"key":"e_1_3_2_1_39_1","first-page":"394","volume-title":"ACM SIGPLAN Conference on Programming Language Design and Implementation, PLDI '12","author":"Leung A.","year":"2012","unstructured":"Leung, A., Gupta, M., Agarwal, Y., Gupta, R., Jhala, R., and Lerner, S. Verifying GPU kernels by test amplification. In ACM SIGPLAN Conference on Programming Language Design and Implementation, PLDI '12, Beijing, China - June 11 - 16, 2012 (2012), J. Vitek, H. Lin, and F. Tip, Eds., ACM, pp. 383\u2013394."},{"key":"e_1_3_2_1_40_1","first-page":"196","volume-title":"Proceedings of the 18th ACM SIGSOFT International Symposium on Foundations of Software Engineering, 2010","author":"Li G.","year":"2010","unstructured":"Li, G., and Gopalakrishnan, G. Scalable smt-based verification of GPU kernel functions. In Proceedings of the 18th ACM SIGSOFT International Symposium on Foundations of Software Engineering, 2010, Santa Fe, NM, USA, November 7-11, 2010 (2010), G. Roman and A. van der Hoek, Eds., ACM, pp. 187\u2013196."},{"key":"e_1_3_2_1_41_1","first-page":"224","volume-title":"Proceedings of the 17th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, PPOPP 2012","author":"Li G.","year":"2012","unstructured":"Li, G., Li, P., Sawaya, G., Gopalakrishnan, G., Ghosh, I., and Rajan, S. P. GKLEE: concolic verification and test generation for gpus. In Proceedings of the 17th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, PPOPP 2012, New Orleans, LA, USA, February 25-29, 2012 (2012), J. Ramanujam and P. Sadayappan, Eds., ACM, pp. 215\u2013224."},{"key":"e_1_3_2_1_42_1","volume-title":"Checkpoint and migration of unix processes in the condor distributed processing system. Tech. rep","author":"Litzkow M.","year":"1997","unstructured":"Litzkow, M., Tannenbaum, T., Basney, J., and Livny, M. Checkpoint and migration of unix processes in the condor distributed processing system. Tech. rep., University of Wisconsin-Madison Department of Computer Sciences, 1997."},{"key":"e_1_3_2_1_43_1","first-page":"172","volume-title":"17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23)","author":"Mai H.","year":"2023","unstructured":"Mai, H., Zhao, J., Zheng, H., Zhao, Y., Liu, Z., Gao, M., Wang, C., Cui, H., Feng, X., and Kozyrakis, C. Honeycomb: Secure and efficient GPU executions via static validation. In 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23) (Boston, MA, 2023), USENIX Association, pp. 155\u2013172."},{"key":"e_1_3_2_1_44_1","volume-title":"https:\/\/github.com\/meta-llama\/llama","author":"Meta","year":"2024","unstructured":"Meta. Llama 2. https:\/\/github.com\/meta-llama\/llama, 2024."},{"key":"e_1_3_2_1_45_1","volume-title":"https:\/\/huggingface.co\/meta-llama\/Llama-3.3-70B-Instruct","author":"Meta","year":"2024","unstructured":"Meta. Llama 3.3. https:\/\/huggingface.co\/meta-llama\/Llama-3.3-70B-Instruct, 2024."},{"key":"e_1_3_2_1_46_1","unstructured":"Microsoft. What runs chatgpt inside microsoft's ai supercomputer. https:\/\/techcommunity.microsoft.com\/blog\/microsoftmechanicsblog\/what-runs-chatgpt-inside-microsofts-ai-supercomputer-featuring-mark-russinovich\/3830281 2025."},{"key":"e_1_3_2_1_47_1","first-page":"216","volume-title":"19th USENIX Conference on File and Storage Technologies, FAST 2021","author":"Mohan J.","year":"2021","unstructured":"Mohan, J., Phanishayee, A., and Chidambaram, V. Checkfreq: Frequent, fine-grained DNN checkpointing. In 19th USENIX Conference on File and Storage Technologies, FAST 2021, February 23-25, 2021 (2021), M. K. Aguilera and G. Yadgar, Eds., USENIX Association, pp. 203\u2013216."},{"key":"e_1_3_2_1_48_1","volume-title":"Creating a communicator. https:\/\/docs.nvidia.com\/deeplearning\/nccl\/user-guide\/docs\/usage\/communicators.html","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. Creating a communicator. https:\/\/docs.nvidia.com\/deeplearning\/nccl\/user-guide\/docs\/usage\/communicators.html, 2024."},{"key":"e_1_3_2_1_49_1","volume-title":"Cuda toolkit 12.3 downloads. https:\/\/developer.nvidia.com\/cuda-12-3-0-download-archive","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. Cuda toolkit 12.3 downloads. https:\/\/developer.nvidia.com\/cuda-12-3-0-download-archive, 2024."},{"key":"e_1_3_2_1_50_1","volume-title":"Cuda toolkit documentation - driver apis. https:\/\/docs.nvidia.com\/cuda\/cuda-driver-api\/index.html","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. Cuda toolkit documentation - driver apis. https:\/\/docs.nvidia.com\/cuda\/cuda-driver-api\/index.html, 2024."},{"key":"e_1_3_2_1_51_1","volume-title":"Implement asynchronous checkpoint saving (with \u2013dist-ckpt-format torch_dist). https:\/\/github.com\/NVIDIA\/Megatron-LM\/commit\/cbb9c05c06b5fa32a8f5b47902751a7bc6d9f112","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. Implement asynchronous checkpoint saving (with \u2013dist-ckpt-format torch_dist). https:\/\/github.com\/NVIDIA\/Megatron-LM\/commit\/cbb9c05c06b5fa32a8f5b47902751a7bc6d9f112, 2024."},{"key":"e_1_3_2_1_52_1","volume-title":"Parallel thread execution isa version 8.4. https:\/\/docs.nvidia.com\/cuda\/parallel-thread-execution\/index.html","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. Parallel thread execution isa version 8.4. https:\/\/docs.nvidia.com\/cuda\/parallel-thread-execution\/index.html, 2024."},{"key":"e_1_3_2_1_53_1","volume-title":"Basic linear algebra on nvidia gpus. https:\/\/developer.nvidia.com\/cublas","author":"NVIDIA.","year":"2025","unstructured":"NVIDIA. Basic linear algebra on nvidia gpus. https:\/\/developer.nvidia.com\/cublas, 2025."},{"key":"e_1_3_2_1_54_1","volume-title":"Context management. https:\/\/docs.nvidia.com\/cuda\/cuda-driver-api\/group__CUDA__CTX.html","author":"NVIDIA.","year":"2025","unstructured":"NVIDIA. Context management. https:\/\/docs.nvidia.com\/cuda\/cuda-driver-api\/group__CUDA__CTX.html, 2025."},{"key":"e_1_3_2_1_55_1","volume-title":"Cuda binary utilities. https:\/\/docs.nvidia.com\/cuda\/cuda-binary-utilities\/index.html","author":"NVIDIA.","year":"2025","unstructured":"NVIDIA. Cuda binary utilities. https:\/\/docs.nvidia.com\/cuda\/cuda-binary-utilities\/index.html, 2025."},{"key":"e_1_3_2_1_56_1","volume-title":"Live migration for gpu-accelerated virtual machines. https:\/\/www.nvidia.com\/en-au\/data-center\/virtualization\/virtual-gpu-migration\/","author":"NVIDIA.","year":"2025","unstructured":"NVIDIA. Live migration for gpu-accelerated virtual machines. https:\/\/www.nvidia.com\/en-au\/data-center\/virtualization\/virtual-gpu-migration\/, 2025."},{"key":"e_1_3_2_1_57_1","volume-title":"https:\/\/github.com\/NVIDIA\/cuda-checkpoint","author":"Nvidia","year":"2025","unstructured":"NVIDIA. Nvidia\/cuda-checkpoint. https:\/\/github.com\/NVIDIA\/cuda-checkpoint, 2025."},{"key":"e_1_3_2_1_58_1","volume-title":"Openai gym. https:\/\/github.com\/openai\/gym","author":"Open AI.","year":"2024","unstructured":"OpenAI. Openai gym. https:\/\/github.com\/openai\/gym, 2024."},{"key":"e_1_3_2_1_59_1","unstructured":"Pytorch. Torchscript."},{"key":"e_1_3_2_1_60_1","volume-title":"Cuda semantics. https:\/\/pytorch.org\/docs\/stable\/notes\/cuda.html#cuda-memory-management","year":"2024","unstructured":"pytorch. Cuda semantics. https:\/\/pytorch.org\/docs\/stable\/notes\/cuda.html#cuda-memory-management, 2024."},{"key":"e_1_3_2_1_61_1","volume-title":"torchvision. https:\/\/github.com\/pytorch\/vision","year":"2024","unstructured":"pytorch. torchvision. https:\/\/github.com\/pytorch\/vision, 2024."},{"key":"e_1_3_2_1_62_1","first-page":"185","volume-title":"Proceedings of the 17th ACM Symposium on Operating System Principles, SOSP 1999, Kiawah Island Resort, near Charleston","author":"Shapiro J. S.","year":"1999","unstructured":"Shapiro, J. S., Smith, J. M., and Farber, D. J. EROS: a fast capability system. In Proceedings of the 17th ACM Symposium on Operating System Principles, SOSP 1999, Kiawah Island Resort, near Charleston, South Carolina, USA, December 12-15, 1999 (1999), D. Kotz and J. Wilkes, Eds., ACM, pp. 170\u2013185."},{"key":"e_1_3_2_1_63_1","first-page":"185","volume-title":"Proceedings of the 17th ACM Symposium on Operating System Principles, SOSP 1999, Kiawah Island Resort, near Charleston","author":"Shapiro J. S.","year":"1999","unstructured":"Shapiro, J. S., Smith, J. M., and Farber, D. J. EROS: a fast capability system. In Proceedings of the 17th ACM Symposium on Operating System Principles, SOSP 1999, Kiawah Island Resort, near Charleston, South Carolina, USA, December 12-15, 1999 (1999), D. Kotz and J. Wilkes, Eds., ACM, pp. 170\u2013185."},{"key":"e_1_3_2_1_64_1","volume-title":"Singularity: Planet-scale, preemptive and elastic scheduling of AI workloads. CoRR abs\/2202.07848","author":"Shukla D.","year":"2022","unstructured":"Shukla, D., Sivathanu, M., Viswanatha, S., Gulavani, B. S., Nehme, R., Agrawal, A., Chen, C., Kwatra, N., Ramjee, R., Sharma, P., Katiyar, A., Modi, V., Sharma, V., Singh, A., Singhal, S., Welankar, K., Xun, L., Anupindi, R., Elangovan, K., Rahman, H., Lin, Z., Seetharaman, R., Xu, C., Ailijiang, E., Krishnappa, S., and Russinovich, M. Singularity: Planet-scale, preemptive and elastic scheduling of AI workloads. CoRR abs\/2202.07848 (2022)."},{"key":"e_1_3_2_1_65_1","volume-title":"Criugpu: Transparent checkpointing of gpu-accelerated workloads. CoRR abs\/2502.16631","author":"Stoyanov R.","year":"2025","unstructured":"Stoyanov, R., Spisakov\u00e1, V., Ramos, J., Gurfinkel, S., Vagin, A., Reber, A., Armour, W., and Bruno, R. Criugpu: Transparent checkpointing of gpu-accelerated workloads. CoRR abs\/2502.16631 (2025)."},{"key":"e_1_3_2_1_66_1","volume-title":"-m. W. Parboil: A revised benchmark suite for scientific and commercial throughput computing","author":"Stratton J. A.","year":"2012","unstructured":"Stratton, J. A., Rodrigues, C., Sung, I.-J., Obeid, N., Chang, L.-W., Anssari, N., Liu, G. D., and Hwu, W.-m. W. Parboil: A revised benchmark suite for scientific and commercial throughput computing. Center for Reliable and High-Performance Computing 127, 7.2 (2012)."},{"key":"e_1_3_2_1_67_1","first-page":"803","volume-title":"Proceedings of the ACM SIGOPS 28th Symposium on Operating Systems Principles (New York, NY, USA, 2021), SOSP '21, Association for Computing Machinery","author":"Tsalapatis E.","unstructured":"Tsalapatis, E., Hancock, R., Barnes, T., and Mashtizadeh, A. J. The aurora single level store operating system. In Proceedings of the ACM SIGOPS 28th Symposium on Operating Systems Principles (New York, NY, USA, 2021), SOSP '21, Association for Computing Machinery, p. 788\u2013803."},{"key":"e_1_3_2_1_68_1","first-page":"572","volume-title":"ASPLOS '21: 26th ACM International Conference on Architectural Support for Programming Languages and Operating Systems","author":"Ustiugov D.","year":"2021","unstructured":"Ustiugov, D., Petrov, P., Kogias, M., Bugnion, E., and Grot, B. Benchmarking, analysis, and optimization of serverless function snapshots. In ASPLOS '21: 26th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Virtual Event, USA, April 19-23, 2021 (2021), T. Sherwood, E. D. Berger, and C. Kozyrakis, Eds., ACM, pp. 559\u2013572."},{"key":"e_1_3_2_1_69_1","volume-title":"Bytecheckpoint: A unified checkpointing system for LLM development. CoRR abs\/2407.20143","author":"Wan B.","year":"2024","unstructured":"Wan, B., Han, M., Sheng, Y., Lai, Z., Zhang, M., Zhang, J., Peng, Y., Lin, H., Liu, X., and Wu, C. Bytecheckpoint: A unified checkpointing system for LLM development. CoRR abs\/2407.20143 (2024)."},{"key":"e_1_3_2_1_70_1","first-page":"16","volume-title":"Proceedings of the Fourteenth EuroSys Conference 2019","author":"Wang K. A.","year":"2019","unstructured":"Wang, K. A., Ho, R., and Wu, P. Replayable execution optimized for page sharing for a managed runtime environment. In Proceedings of the Fourteenth EuroSys Conference 2019, Dresden, Germany, March 25-28, 2019 (2019), G. Candea, R. van Renesse, and C. Fetzer, Eds., ACM, pp. 39:1\u201339:16."},{"key":"e_1_3_2_1_71_1","volume-title":"Characterizing network requirements for GPU API remoting in AI applications. CoRR abs\/2401.13354","author":"Wang T.","year":"2024","unstructured":"Wang, T., Chen, Z., Wei, X., Gu, J., Chen, R., and Chen, H. Characterizing network requirements for GPU API remoting in AI applications. CoRR abs\/2401.13354 (2024)."},{"key":"e_1_3_2_1_72_1","first-page":"381","volume-title":"Proceedings of the 29th Symposium on Operating Systems Principles, SOSP 2023","author":"Wang Z.","year":"2023","unstructured":"Wang, Z., Jia, Z., Zheng, S., Zhang, Z., Fu, X., Ng, T. S. E., and Wang, Y. GEMINI: fast failure recovery in distributed training with in-memory checkpoints. In Proceedings of the 29th Symposium on Operating Systems Principles, SOSP 2023, Koblenz, Germany, October 23-26, 2023 (2023), J. Flinn, M. I. Seltzer, P. Druschel, A. Kaufmann, and J. Mace, Eds., ACM, pp. 364\u2013381."},{"key":"e_1_3_2_1_73_1","first-page":"517","volume-title":"17th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2023","author":"Wei X.","year":"2023","unstructured":"Wei, X., Lu, F., Wang, T., Gu, J., Yang, Y., Chen, R., and Chen, H. No provisioned concurrency: Fast rdma-codesigned remote fork for serverless computing. In 17th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2023, Boston, MA, USA, July 10-12, 2023 (2023), R. Geambasu and E. Nightingale, Eds., USENIX Association, pp. 497\u2013517."},{"key":"e_1_3_2_1_74_1","first-page":"16","volume-title":"Proceedings of the 29th Symposium on Operating Systems Principles, SOSP 2023","author":"Wu F.","year":"2023","unstructured":"Wu, F., Dong, M., Mo, G., and Chen, H. Treesls: A whole-system persistent microkernel with tree-structured state checkpoint on NVM. In Proceedings of the 29th Symposium on Operating Systems Principles, SOSP 2023, Koblenz, Germany, October 23-26, 2023 (2023), J. Flinn, M. I. Seltzer, P. Druschel, A. Kaufmann, and J. Mace, Eds., ACM, pp. 1\u201316."},{"key":"e_1_3_2_1_75_1","first-page":"610","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2018","author":"Xiao W.","year":"2018","unstructured":"Xiao, W., Bhardwaj, R., Ramjee, R., Sivathanu, M., Kwatra, N., Han, Z., Patel, P., Peng, X., Zhao, H., Zhang, Q., Yang, F., and Zhou, L. Gandiva: Introspective cluster scheduling for deep learning. In 13th USENIX Symposium on Operating Systems Design and Implementation, OSDI 2018, Carlsbad, CA, USA, October 8-10, 2018 (2018), A. C. Arpaci-Dusseau and G. Voelker, Eds., USENIX Association, pp. 595\u2013610."},{"key":"e_1_3_2_1_76_1","volume-title":"Proceedings of the 2024 ACM Symposium on Cloud Computing","author":"Yang Y.","year":"2024","unstructured":"Yang, Y., Du, D., Song, H., and Xia, Y. On-demand and parallel checkpoint\/restore for gpu applications. In Proceedings of the 2024 ACM Symposium on Cloud Computing (New York, NY, USA, 2024), SoCC '24, Association for Computing Machinery, p. 415\u2013433."},{"key":"e_1_3_2_1_77_1","volume-title":"Faaswap: Slo-aware, gpu-efficient serverless inference via model swapping. CoRR abs\/2306.03622","author":"Yu M.","year":"2023","unstructured":"Yu, M., Wang, A., Chen, D., Yu, H., Luo, X., Li, Z., Wang, W., Chen, R., Nie, D., and Yang, H. Faaswap: Slo-aware, gpu-efficient serverless inference via model swapping. CoRR abs\/2306.03622 (2023)."},{"key":"e_1_3_2_1_78_1","first-page":"668","volume-title":"Proceedings of the 30th ACM International Conference on Architectural Support for Programming Languages and Operating Systems","volume":"1","author":"Zeng S.","year":"2025","unstructured":"Zeng, S., Xie, M., Gao, S., Chen, Y., and Lu, Y. Medusa: Accelerating serverless LLM inference with materialization. In Proceedings of the 30th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 1, ASP-LOS 2025, Rotterdam, The Netherlands, 30 March 2025 - 3 April 2025 (2025), L. Eeckhout, G. Smaragdakis, K. Liang, A. Sampson, M. A. Kim, and C. J. Rossbach, Eds., ACM, pp. 653\u2013668."},{"key":"e_1_3_2_1_79_1","volume-title":"OPT: open pre-trained transformer language models. CoRR abs\/2205.01068","author":"Zhang S.","year":"2022","unstructured":"Zhang, S., Roller, S., Goyal, N., Artetxe, M., Chen, M., Chen, S., Dewan, C., Diab, M. T., Li, X., Lin, X. V., Mihaylov, T., Ott, M., Shleifer, S., Shuster, K., Simig, D., Koura, P. S., Sridhar, A., Wang, T., and Zettlemoyer, L. OPT: open pre-trained transformer language models. CoRR abs\/2205.01068 (2022)."}],"event":{"name":"SOSP '25: ACM SIGOPS 31st Symposium on Operating Systems Principles","location":"Lotte Hotel World Seoul Republic of Korea","acronym":"SOSP '25","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems","USENIX"]},"container-title":["Proceedings of the ACM SIGOPS 31st Symposium on Operating Systems Principles"],"original-title":[],"deposited":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T12:52:23Z","timestamp":1759323143000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731569.3764813"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,12]]},"references-count":79,"alternative-id":["10.1145\/3731569.3764813","10.1145\/3731569"],"URL":"https:\/\/doi.org\/10.1145\/3731569.3764813","relation":{},"subject":[],"published":{"date-parts":[[2025,10,12]]},"assertion":[{"value":"2025-10-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}