{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T13:06:29Z","timestamp":1780664789176,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":106,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,26]],"date-time":"2026-04-26T00:00:00Z","timestamp":1777161600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62432010"],"award-info":[{"award-number":["62432010"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62272291"],"award-info":[{"award-number":["62272291"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Fundamental and Interdisciplinary Dis- ciplines Breakthrough Plan of the Ministry of Education of China","award":["JYB2025XDXM122"],"award-info":[{"award-number":["JYB2025XDXM122"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,27]]},"DOI":"10.1145\/3767295.3803597","type":"proceedings-article","created":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T20:20:04Z","timestamp":1777062004000},"page":"852-971","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Accurate and Ultra-Fast Launch-Time Validation of Idempotency for GPU Kernels"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-1536-7485","authenticated-orcid":false,"given":"Mingcong","family":"Han","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2568-177X","authenticated-orcid":false,"given":"Weihang","family":"Shen","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6115-8130","authenticated-orcid":false,"given":"Rong","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9720-0361","authenticated-orcid":false,"given":"Haibo","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai JiaoTong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,26]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"PyTorch Github Repository. https:\/\/github.com\/pytorch\/pytorch\/blob\/664058fa83f1d8eede5d66418abff6e20bd76ca8."},{"key":"e_1_3_2_1_2_1","unstructured":"Rodinia Website. https:\/\/rodinia.cs.virginia.edu\/doku.php."},{"key":"e_1_3_2_1_3_1","unstructured":"TensorFlow Pre-trained MobileNetV1 Models. https:\/\/github.com\/tensorflow\/models\/blob\/505f9bf352d35700bf2596f7d9ce908881f81a07\/research\/slim\/nets\/mobilenet_v1.md."},{"key":"e_1_3_2_1_4_1","unstructured":"TorchVision. https:\/\/pytorch.org\/vision\/stable\/models.html."},{"key":"e_1_3_2_1_5_1","unstructured":"Induction Variables pages 35\u201358. Springer US Boston MA 1995."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/RTAS48715.2020.000-1"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-63387-9_25"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10703-021-00362-8"},{"key":"e_1_3_2_1_9_1","first-page":"308","volume-title":"2024 57th IEEE\/ACM International Symposium on Microarchitecture (MICRO)","author":"Pratheek","year":"2024","unstructured":"Pratheek B., Guilherme Cox, J\u00e1n Vesel\u00fd, and Arkaprava Basu. Suv: Static analysis guided unified virtual memory. 2024 57th IEEE\/ACM International Symposium on Microarchitecture (MICRO), pages 293\u2013308, 2024."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-08867-9_15"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/2743017"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/2384616.2384625"},{"key":"e_1_3_2_1_13_1","volume-title":"Third Workshop on Software Tools for MultiCore Systems","volume":"33","author":"Boyer Michael","year":"2008","unstructured":"Michael Boyer, Kevin Skadron, and Westley Weimer. Automated dynamic analysis of cuda programs. In Third Workshop on Software Tools for MultiCore Systems, volume 33, 2008."},{"key":"e_1_3_2_1_14_1","first-page":"54","volume-title":"IEEE International Symposium on Workload Characterization (IISWC)","author":"Che Shuai","year":"2009","unstructured":"Shuai Che, Michael Boyer, Jiayuan Meng, David Tarjan, Jeremy W. Sheaffer, Sang ha Lee, and Kevin Skadron. Rodinia: A benchmark suite for heterogeneous computing. IEEE International Symposium on Workload Characterization (IISWC), pages 44\u201354, 2009."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2063384.2063401"},{"key":"e_1_3_2_1_16_1","volume-title":"cuDNN: Efficient primitives for deep learning. ArXiv, abs\/1410.0759","author":"Chetlur Sharan","year":"2014","unstructured":"Sharan Chetlur, Cliff Woolley, Philippe Vandermersch, Jonathan M. Cohen, John Tran, Bryan Catanzaro, and Evan Shelhamer. cuDNN: Efficient primitives for deep learning. ArXiv, abs\/1410.0759, 2014."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-38088-4_15"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSE.2013.2297120"},{"key":"e_1_3_2_1_19_1","volume-title":"The discussion of inplace update in dataflow block. https:\/\/discuss.tvm.apache.org\/t\/discuss-inplace-update-in-dataflow-block\/14669","author":"Community TVM","year":"2023","unstructured":"TVM Community. The discussion of inplace update in dataflow block. https:\/\/discuss.tvm.apache.org\/t\/discuss-inplace-update-in-dataflow-block\/14669, 2023."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1088\/1741-2552\/ab0ab5"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/2155620.2155637"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/2254064.2254120"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-78800-3_24"},{"key":"e_1_3_2_1_24_1","volume-title":"https:\/\/github.com\/NVIDIA\/cutlass","author":"Cutlass CUTLASS","year":"2021","unstructured":"CUTLASS developers. Cutlass. https:\/\/github.com\/NVIDIA\/cutlass, 2021."},{"key":"e_1_3_2_1_25_1","first-page":"910","volume-title":"17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23)","author":"Ding Haoran","year":"2023","unstructured":"Haoran Ding, Zhaoguo Wang, Zhuohao Shen, Rong Chen, and Haibo Chen. Automated verification of idempotence for stateful serverless applications. In 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23), pages 887\u2013910, Boston, MA, 2023. USENIX Association."},{"key":"e_1_3_2_1_26_1","first-page":"943","volume-title":"19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22)","author":"Eisenman Assaf","year":"2022","unstructured":"Assaf Eisenman, Kiran Kumar Matam, Steven Ingram, Dheevatsa Mudigere, Raghuraman Krishnamoorthi, Krishnakumar Nair, Misha Smelyanskiy, and Murali Annavaram. Check-N-Run: a checkpointing system for training deep learning recommendation models. In 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22), pages 929\u2013943, Renton, WA, April 2022. USENIX Association."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2017.7863729"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41591-018-0316-z"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575697"},{"key":"e_1_3_2_1_30_1","volume-title":"Symposium on Networked Systems Design and Implementation","author":"Go Younghwan","year":"2017","unstructured":"Younghwan Go, Muhammad Asim Jamshed, Younggyoun Moon, Changho Hwang, and KyoungSoo Park. Apunet: Revitalizing gpu as packet processing accelerator. In Symposium on Networked Systems Design and Implementation, 2017."},{"key":"e_1_3_2_1_31_1","first-page":"4050","volume-title":"33rd USENIX Security Symposium (USENIX Security 24)","author":"Guo Yanan","year":"2024","unstructured":"Yanan Guo, Zhenkai Zhang, and Jun Yang. GPU memory exploitation for fun and profit. In 33rd USENIX Security Symposium (USENIX Security 24), pages 4033\u20134050, Philadelphia, PA, August 2024. USENIX Association."},{"key":"e_1_3_2_1_32_1","first-page":"558","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Han Mingcong","year":"2022","unstructured":"Mingcong Han, Hanze Zhang, Rong Chen, and Haibo Chen. Microsecond-scale preemption for concurrent GPU-accelerated DNN inferences. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22), pages 539\u2013558, Carlsbad, CA, July 2022. USENIX Association."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2019.8661186"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589105"},{"issue":"56","key":"e_1_3_2_1_36_1","first-page":"65","article-title":"Idempotence is not a medical condition","volume":"55","author":"Helland Pat","year":"2012","unstructured":"Pat Helland. Idempotence is not a medical condition. Communications of the ACM, 55:56 - 65, 2012.","journal-title":"Communications of the ACM"},{"key":"e_1_3_2_1_37_1","volume-title":"Deep neural networks for acoustic modeling in speech recognition: The shared views of four research groups","author":"Hinton Geoffrey","year":"2012","unstructured":"Geoffrey Hinton, Li Deng, Dong Yu, George E Dahl, Abdelrahman Mohamed, Navdeep Jaitly, Andrew Senior, Vincent Vanhoucke, Patrick Nguyen, Tara N Sainath, et al. Deep neural networks for acoustic modeling in speech recognition: The shared views of four research groups. IEEE Signal processing magazine, 29(6):82\u201397, 2012."},{"key":"e_1_3_2_1_38_1","volume-title":"Mobilenets: Efficient convolutional neural networks for mobile vision applications. ArXiv, abs\/1704.04861","author":"Howard Andrew G.","year":"2017","unstructured":"Andrew G. Howard, Menglong Zhu, Bo Chen, Dmitry Kalenichenko, Weijun Wang, Tobias Weyand, Marco Andreetto, and Hartwig Adam. Mobilenets: Efficient convolutional neural networks for mobile vision applications. ArXiv, abs\/1704.04861, 2017."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2010.107"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/RTSS49844.2020.00027"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3360575"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3552326.3567508"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477132.3483541"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3332466.3374531"},{"key":"e_1_3_2_1_46_1","volume-title":"Symposium on Networked Systems Design and Implementation","author":"Kalia Anuj","year":"2015","unstructured":"Anuj Kalia, Dong Zhou, Michael Kaminsky, and David G. Andersen. Raising the bar for using gpus in software packet processing. In Symposium on Networked Systems Design and Implementation, 2015."},{"key":"e_1_3_2_1_47_1","volume-title":"Proceedings of the ACM SIGOPS 28th Symposium on Operating Systems Principles","author":"Aditya","year":"2021","unstructured":"Aditya K. Kamath and Arkaprava Basu. iguard: In-gpu advanced race detection. Proceedings of the ACM SIGOPS 28th Symposium on Operating Systems Principles, 2021."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/1152649.1152653"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2020.2988251"},{"key":"e_1_3_2_1_50_1","volume-title":"Tesla's autonomy event:Impressive progress with an unrealistic timeline. https:\/\/arstechnica.com\/cars\/2019\/04\/teslas-autonomy-event-impressive-progress-with-an-unrealistic-timeline\/","author":"TIMOTHY B.","year":"2019","unstructured":"TIMOTHY B. LEE. Tesla's autonomy event:Impressive progress with an unrealistic timeline. https:\/\/arstechnica.com\/cars\/2019\/04\/teslas-autonomy-event-impressive-progress-with-an-unrealistic-timeline\/, 2019."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00014"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/2254064.2254110"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/1882291.1882320"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/2145816.2145844"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2012.94"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2014.20"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2019.00032"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2016.76"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/2930667"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.996"},{"key":"e_1_3_2_1_61_1","first-page":"172","volume-title":"17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23)","author":"Mai HaoHui","year":"2023","unstructured":"HaoHui Mai, Jiacheng Zhao, Hongren Zheng, Yiyang Zhao, Zibin Liu, Mingyu Gao, Cong Wang, Huimin Cui, Xiaobing Feng, and Christos Kozyrakis. Honeycomb: Secure and efficient GPU executions via static validation. In 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23), pages 155\u2013172, Boston, MA, 2023. USENIX Association."},{"key":"e_1_3_2_1_62_1","first-page":"216","volume-title":"19th USENIX Conference on File and Storage Technologies (FAST 21)","author":"Mohan Jayashree","unstructured":"Jayashree Mohan, Amar Phanishayee, and Vijay Chidambaram. CheckFreq: Frequent, Fine-Grained DNN checkpointing. In 19th USENIX Conference on File and Storage Technologies (FAST 21), pages 203\u2013216. USENIX Association, February 2021."},{"key":"e_1_3_2_1_63_1","unstructured":"NVIDIA. NVIDIA GPU Instruction Set Reference. https:\/\/docs.nvidia.com\/cuda\/cuda-binary-utilities\/index.html#instruction-set-reference."},{"key":"e_1_3_2_1_64_1","unstructured":"NVIDIA. NVIDIA TensorRT. https:\/\/developer.nvidia.com\/tensorrt."},{"key":"e_1_3_2_1_65_1","volume-title":"NVIDIA Tesla V100 GPU Architecture. https:\/\/images.nvidia.cn\/content\/volta-architecture\/pdf\/volta-architecture-whitepaper.pdf","author":"NVIDIA.","year":"2017","unstructured":"NVIDIA. NVIDIA Tesla V100 GPU Architecture. https:\/\/images.nvidia.cn\/content\/volta-architecture\/pdf\/volta-architecture-whitepaper.pdf, 2017."},{"key":"e_1_3_2_1_66_1","volume-title":"https:\/\/github.com\/NVIDIA\/cuda-samples","author":"Samples NVIDIA. CUDA","year":"2023","unstructured":"NVIDIA. CUDA Samples. https:\/\/github.com\/NVIDIA\/cuda-samples, 2023."},{"key":"e_1_3_2_1_67_1","volume-title":"https:\/\/github.com\/NVIDIA\/FasterTransformer","author":"FasterTransformer NVIDIA.","year":"2023","unstructured":"NVIDIA. FasterTransformer. https:\/\/github.com\/NVIDIA\/FasterTransformer, 2023."},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507758"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid49817.2020.00-69"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1145\/2694344.2694346"},{"key":"e_1_3_2_1_71_1","volume-title":"NeurIPS","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, S. Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, N. Gimelshein, L. Antiga, Alban Desmaison, Andreas K\u00f6pf, Edward Yang, Zach DeVito, Martin Raison, Alykhan Tejani, Sasank Chilamkurthy, Benoit Steiner, Lu Fang, Junjie Bai, and Soumith Chintala. Pytorch: An imperative style, high-performance deep learning library. In NeurIPS, 2019."},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1109\/JETCAS.2021.3121259"},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613154"},{"key":"e_1_3_2_1_74_1","volume-title":"Language models are unsupervised multitask learners","author":"Radford Alec","year":"2019","unstructured":"Alec Radford, Jeff Wu, Rewon Child, David Luan, Dario Amodei, and Ilya Sutskever. Language models are unsupervised multitask learners. 2019."},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.1145\/2429069.2429100"},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2024.104879"},{"key":"e_1_3_2_1_77_1","volume-title":"Towards fully-fledged gpu multitasking via proactive memory scheduling","author":"Shen Weihang","year":"2026","unstructured":"Weihang Shen, Yinqiu Chen, Rong Chen, and Haibo Chen. Towards fully-fledged gpu multitasking via proactive memory scheduling, 2026."},{"key":"e_1_3_2_1_78_1","first-page":"692","volume-title":"Haibo Chen. XSched: Preemptive Scheduling for Diverse XPUs. In 19th USENIX Symposium on Operating Systems Design and Implementation (OSDI 25)","author":"Shen Weihang","year":"2025","unstructured":"Weihang Shen, Mingcong Han, Jialong Liu, Rong Chen, and Haibo Chen. XSched: Preemptive Scheduling for Diverse XPUs. In 19th USENIX Symposium on Operating Systems Design and Implementation (OSDI 25), pages 671\u2013692, Boston, MA, July 2025. USENIX Association."},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2016.17"},{"key":"e_1_3_2_1_80_1","volume-title":"Planet-scale, preemptive and elastic scheduling of AI workloads. CoRR, abs\/2202.07848","author":"Shukla Dharma","year":"2022","unstructured":"Dharma Shukla, Muthian Sivathanu, Srinidhi Viswanatha, Bhargav S. Gulavani, Rimma Nehme, Amey Agrawal, Chen Chen, Nipun Kwatra, Ramachandran Ramjee, Pankaj Sharma, Atul Katiyar, Vipul Modi, Vaibhav Sharma, Abhishek Singh, Shreshth Singhal, Kaustubh Welankar, Lu Xun, Ravi Anupindi, Karthik Elangovan, Hasibur Rahman, Zhou Lin, Rahul Seetharaman, Cheng Xu, Eddie Ailijiang, Suresh Krishnappa, and Mark Russinovich. Singularity: Planet-scale, preemptive and elastic scheduling of AI workloads. CoRR, abs\/2202.07848, 2022."},{"key":"e_1_3_2_1_81_1","volume-title":"USENIX Annual Technical Conference","author":"Sigurbjarnarson Helgi","year":"2016","unstructured":"Helgi Sigurbjarnarson, James Bornholt, Nicolas Christin, and Lorrie Faith Cranor. Push-button verification of file systems via crash refinement. In USENIX Annual Technical Conference, 2016."},{"key":"e_1_3_2_1_82_1","volume-title":"TOCS","author":"Silberstein Mark","year":"2013","unstructured":"Mark Silberstein, Bryan Ford, Idit Keidar, and Emmett Witchel. Gpufs: Integrating a file system with gpus. In TOCS, 2013."},{"key":"e_1_3_2_1_83_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556, 2014."},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"publisher","DOI":"10.1145\/3342195.3387535"},{"key":"e_1_3_2_1_85_1","volume-title":"Parboil: A revised benchmark suite for scientific and commercial throughput computing","author":"Stratton John A.","year":"2012","unstructured":"John A. Stratton, Christopher I. Rodrigues, I-Jui Sung, Nady Obeid, Li-Wen Chang, Nasser Anssari, Geng Liu, and Wen mei W. Hwu. Parboil: A revised benchmark suite for scientific and commercial throughput computing. 2012."},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"publisher","DOI":"10.1145\/3360609"},{"key":"e_1_3_2_1_87_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"key":"e_1_3_2_1_88_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2014.6853208"},{"key":"e_1_3_2_1_89_1","volume-title":"Gemini: A family of highly capable multimodal models","author":"Team Gemini","year":"2025","unstructured":"Gemini Team. Gemini: A family of highly capable multimodal models, 2025."},{"key":"e_1_3_2_1_90_1","volume-title":"The llama 3 herd of models","author":"Team Llama","year":"2024","unstructured":"Llama Team. The llama 3 herd of models, 2024."},{"key":"e_1_3_2_1_91_1","volume-title":"GPUs and accelerators. https:\/\/tvm.apache.org\/","author":"Apache","year":"2021","unstructured":"Apache TVM. Apache TVM: An End to End Machine Learning Compiler Framework for CPUs, GPUs and accelerators. https:\/\/tvm.apache.org\/, 2021."},{"key":"e_1_3_2_1_92_1","first-page":"48","volume-title":"Proceeding of the 41st Annual International Symposium on Computer Architecuture, ISCA '14","author":"Upasani Gaurang","unstructured":"Gaurang Upasani, Xavier Vera, and Antonio Gonz\u00e1lez. Avoiding core's due & sdc via acoustic wave detectors and tailored error containment and recovery. In Proceeding of the 41st Annual International Symposium on Computer Architecuture, ISCA '14, page 37\u201348. IEEE Press, 2014."},{"key":"e_1_3_2_1_93_1","volume-title":"USENIX Symposium on Operating Systems Design and Implementation","author":"van der Woude Joel","year":"2016","unstructured":"Joel van der Woude and Matthew Hicks. Intermittent computation without hardware support or programmer intervention. In USENIX Symposium on Operating Systems Design and Implementation, 2016."},{"key":"e_1_3_2_1_94_1","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358307"},{"key":"e_1_3_2_1_95_1","volume-title":"et al. Tacotron: Afully end-to-end text-to-speech synthesis model. arXiv preprint arXiv:1703.10135","author":"Wang Yuxuan","year":"2017","unstructured":"Yuxuan Wang, RJ Skerry-Ryan, Daisy Stanton, Yonghui Wu, Ron J Weiss, Navdeep Jaitly, Zongheng Yang, Ying Xiao, Zhifeng Chen, Samy Bengio, et al. Tacotron: Afully end-to-end text-to-speech synthesis model. arXiv preprint arXiv:1703.10135, 2017."},{"key":"e_1_3_2_1_96_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613145"},{"key":"e_1_3_2_1_97_1","doi-asserted-by":"publisher","DOI":"10.1145\/3731569.3764813"},{"key":"e_1_3_2_1_98_1","first-page":"610","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation","author":"Xiao Wencong","year":"2018","unstructured":"Wencong Xiao, Romil Bhardwaj, Ramachandran Ramjee, Muthian Sivathanu, Nipun Kwatra, Zhenhua Han, Pratyush Patel, Xuan Peng, Hanyu Zhao, Quanlu Zhang, Fan Yang, and Lidong Zhou. Gandiva: Introspective cluster scheduling for deep learning. In 13th USENIX Symposium on Operating Systems Design and Implementation, pages 595\u2013610, Carlsbad, CA, October 2018. USENIX Association."},{"key":"e_1_3_2_1_99_1","doi-asserted-by":"publisher","DOI":"10.1145\/3372224.3419192"},{"key":"e_1_3_2_1_100_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC50251.2020.00032"},{"key":"e_1_3_2_1_101_1","volume-title":"USENIX Symposium on Operating Systems Design and Implementation","author":"Zhang Haoran","year":"2020","unstructured":"Haoran Zhang, Adney Cardoza, Peter Baile Chen, Sebastian Angel, and Vincent Liu. Fault-tolerant and transactional stateful serverless workflows. In USENIX Symposium on Operating Systems Design and Implementation, 2020."},{"key":"e_1_3_2_1_102_1","unstructured":"Susan Zhang Stephen Roller Naman Goyal Mikel Artetxe Moya Chen Shuohui Chen Christopher Dewan Mona T. Diab Xian Li Xi Victoria Lin Todor Mihaylov Myle Ott Sam Shleifer Kurt Shuster Daniel Simig Punit Singh Koura Anjali Sridhar Tianlu Wang and Luke Zettlemoyer. Opt: Open pre-trained transformer language models. ArXiv abs\/2205.01068 2022."},{"key":"e_1_3_2_1_103_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO56248.2022.00030"},{"key":"e_1_3_2_1_104_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2013.44"},{"key":"e_1_3_2_1_105_1","volume-title":"Proceedings of the 2025 USENIX Conference on Usenix Annual Technical Conference, USENIX ATC '25, USA","author":"Zheng Wenxin","year":"2025","unstructured":"Wenxin Zheng, Bin Xu, Jinyu Gu, and Haibo Chen. Save: software-implemented fault tolerance for model inference against gpu memory bit flips. In Proceedings of the 2025 USENIX Conference on Usenix Annual Technical Conference, USENIX ATC '25, USA, 2025. USENIX Association."},{"key":"e_1_3_2_1_106_1","first-page":"286","volume-title":"17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23)","author":"Zhuang Siyuan","year":"2023","unstructured":"Siyuan Zhuang, Stephanie Wang, Eric Liang, Yi Cheng, and Ion Stoica. ExoFlow: A universal workflow system for Exactly-Once DAGs. In 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23), pages 269\u2013286, Boston, MA, 2023. USENIX Association."}],"event":{"name":"EUROSYS '26: 21st European Conference on Computer Systems","location":"McEwan Hall\/The University of Edinburgh Edinburgh Scotland UK","acronym":"EUROSYS '26","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 21st European Conference on Computer Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3767295.3803597","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T12:19:26Z","timestamp":1780661966000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3767295.3803597"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,26]]},"references-count":106,"alternative-id":["10.1145\/3767295.3803597","10.1145\/3767295"],"URL":"https:\/\/doi.org\/10.1145\/3767295.3803597","relation":{},"subject":[],"published":{"date-parts":[[2026,4,26]]},"assertion":[{"value":"2026-04-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}