{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T07:14:24Z","timestamp":1772694864667,"version":"3.50.1"},"reference-count":71,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,1,31]],"date-time":"2026-01-31T00:00:00Z","timestamp":1769817600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,31]],"date-time":"2026-01-31T00:00:00Z","timestamp":1769817600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000028","name":"Semiconductor Research Corporation","doi-asserted-by":"publisher","award":["3281.001"],"award-info":[{"award-number":["3281.001"]}],"id":[{"id":"10.13039\/100000028","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,1,31]]},"DOI":"10.1109\/hpca68181.2026.11408597","type":"proceedings-article","created":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T20:47:22Z","timestamp":1772657242000},"page":"1-15","source":"Crossref","is-referenced-by-count":0,"title":["Exploration of LLM Workload Reliability Based on di\/dt Effects and Voltage Droops"],"prefix":"10.1109","author":[{"given":"Zhixing","family":"Jiang","sequence":"first","affiliation":[{"name":"University of Texas-Austin"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Justin","family":"Garrigus","sequence":"additional","affiliation":[{"name":"University of Texas-Austin"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Allison","family":"Seigler","sequence":"additional","affiliation":[{"name":"University of Texas-Austin"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ethan","family":"Syed","sequence":"additional","affiliation":[{"name":"University of Texas-Austin"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yan-Lun","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Texas-Austin"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mehdi","family":"Sadi","sequence":"additional","affiliation":[{"name":"Advanced Micro Devices, Inc."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tawfik","family":"Rahal-Arabi","sequence":"additional","affiliation":[{"name":"Advanced Micro Devices, Inc."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lizy Kurian","family":"John","sequence":"additional","affiliation":[{"name":"University of Texas-Austin"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1147\/rd.475.0653"},{"key":"ref2","volume-title":"Power stabilization for ai training datacenters","author":"Choukse","year":"2025"},{"key":"ref3","volume-title":"NVIDIA A100 Tensor Core GPU Architecture,2020, architecture analysis; includes MIG, tensor cores, HBM2, NVLink"},{"key":"ref4","volume-title":"Deepseek-rl: Incentivizing reasoning capability in llms via reinforcement learning","author":"Guo","year":"2025"},{"key":"ref5","first-page":"505","article-title":"Minder: Faulty machine detection for large-scale distributed model training","volume-title":"22nd USENIX Symposium on Networked Systems Design and Implementation (NSDI 25)","author":"Deng"},{"key":"ref6","volume-title":"Detecting silent data corruptions in the wild","author":"Dixit","year":"2022"},{"key":"ref7","volume-title":"Silent data corruptions at scale","author":"Dixit","year":"2021"},{"key":"ref8","first-page":"332","article-title":"Mitigating inductive noise in smt processors","volume-title":"Proceedings of the 2004 International Symposium on Low Power Electronics and Design (IEEE Cat. No.04TH8758)","author":"El-Essawy","year":"2004"},{"key":"ref9","first-page":"2171","article-title":"DEAP: Evolutionary algorithms made easy","volume":"13","author":"Fortin","year":"2012","journal-title":"Journal of Machine Learning Research"},{"key":"ref10","first-page":"19","article-title":"System-level max power (sympo) - a systematic approach for escalating system-level power consumption using synthetic benchmarks","volume-title":"2010 19th International Conference on Parallel Architectures and Compilation Techniques (PACT)","author":"Ganesan","year":"2010"},{"key":"ref11","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/2063384.2063455","article-title":"Maximum multicore power (mampo) an automatic multithreaded synthetic power virus generation framework for multicore systems","volume-title":"SC \u201911: Proceedings of 2011 International Conference for High Performance Computing, Networking, Storage and Analysis","author":"Ganesan","year":"2011"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/DATE.2007.364663"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2008.4658654"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/DATE.2009.5090651"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/1283780.1283808"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3458336.3465297"},{"key":"ref17","article-title":"Characterization of large language model development in the datacenter","volume-title":"Proceedings of the 21st USENIX Symposium on Networked Systems Design and Implementation","author":"Hu"},{"key":"ref18","first-page":"947","article-title":"Analysis of Large-Scale Multi-Tenant GPU clusters for DNN training workloads","volume-title":"2019 USENIX Annual Technical Conference (USENIX ATC 19)","author":"Jeon"},{"key":"ref19","first-page":"202","article-title":"Preventing the immense increase in the life-cycle energy and carbon footprints of LLM-powered intelligent chatbots","volume":"40","author":"Jiang","year":"2024","journal-title":"PII"},{"key":"ref20","article-title":"Megascale: scaling large language model training to more than 10,000 gpus","volume-title":"Proceedings of the 21st USENIX Symposium on Networked Systems Design and Implementation, ser. NSDI\u201924. USA: USENIX Association","author":"Jiang","year":"2024"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2003.1183526"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2008.4658642"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ASYNC.2016.13"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/3466752.3480063"},{"key":"ref25","article-title":"Measuring code optimization impact on voltage noise","volume-title":"Proceedings of the 9th Silicon Errors in Logic - System Effects Workshop (SELSE 9)","author":"Kanev","year":"2013"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA45697.2020.00047"},{"key":"ref27","volume-title":"Full stack optimization of transformer inference: a survey","author":"Kim","year":"2023"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ISLPED.2011.5993645"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2012.28"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1145\/2830772.2830811"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2015.7056030"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/2627369.2627605"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ITC50671.2022.00076"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3623773"},{"key":"ref35","first-page":"15","article-title":"Hostping: Diagnosing intra-host network bottlenecks in RDMA servers","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Liu"},{"key":"ref36","first-page":"637651","article-title":"Understanding and improving failure tolerant training for deep learning recommendation with partial recovery","volume-title":"Proceedings of Machine Learning and Systems","volume":"3","author":"Maeng","year":"2021"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2012.6237022"},{"key":"ref38","volume-title":"CUDA C Best Practices Guide","year":"2016"},{"key":"ref39","article-title":"Nvidia volta: The world\u2019s most advanced data center gpu","volume-title":"NVIDIA Corporation","year":"2017"},{"key":"ref40","article-title":"Nvidia ampere architecture: Whitepaper","volume-title":"NVIDIA Corporation","year":"2020"},{"key":"ref41","article-title":"Nvidia h100 tensor core gpu: Hopper architecture whitepaper","volume-title":"NVIDIA Corporation","year":"2022"},{"key":"ref42","article-title":"CUTLASS: Cuda templates for linear algebra subroutines","volume-title":"NVIDIA Corporation","year":"2024"},{"key":"ref43","volume-title":"Gpt-4 technical report","author":"Achiam","year":"2024"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00045"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1145\/2228360.2228571"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/LPE.2003.1231866"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2004.1310782"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/APCCAS62602.2024.10808935"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1145\/1839667.1839674"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2010.25"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2009.4798233"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2010.35"},{"key":"ref53","volume-title":"Silifuzz: Fuzzing cpus by proxy","author":"Serebryany","year":"2021"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00087"},{"key":"ref55","doi-asserted-by":"crossref","first-page":"74","DOI":"10.1016\/j.vlsi.2017.02.002","article-title":"Scaling equations for the accurate prediction of CMOS device performance from 180 nm to 7 nm","volume":"58","author":"Stillmaker","year":"2017","journal-title":"Integration"},{"key":"ref56","volume-title":"Summarizing cpu and gpu design trends with product data","author":"Sun","year":"2020"},{"key":"ref57","first-page":"197","article-title":"Mgpusim: Enabling multi-gpu performance modeling and optimization","volume-title":"2019 ACM\/ IEEE 46th Annual International Symposium on Computer Architecture (ISCA)","author":"Sun","year":"2019"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/DSN48987.2021.00043"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2023.3279304"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1145\/2967938.2967951"},{"key":"ref61","volume-title":"Gemma: Open models based on gemini research and technology","author":"Team","year":"2024"},{"key":"ref62","first-page":"298245","article-title":"Intel\u00ae pentium\u00ae 4 processor in the 423-pin package \/ intel\u00ae 850 chipset platform design guide","author":"Team","year":"2002","journal-title":"Intel Corporation, Tech. Rep."},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2016.7446061"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2016.7482076"},{"key":"ref65","volume-title":"Bytecheckpoint: A unified checkpointing system for large foundation model development","author":"Wan","year":"2025"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613149"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/JSSC.2006.870925"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/TCSII.2015.2391632"},{"key":"ref69","first-page":"523","article-title":"Holmes: Localizing irregularities in LLM training with megascale GPU clusters","volume-title":"22nd USENIX Symposium on Networked Systems Design and Implementation (NSDI 25)","author":"Yao"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1145\/3377811.3380362"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2018.00039"}],"event":{"name":"2026 IEEE International Symposium on High Performance Computer Architecture (HPCA)","location":"Sydney, Australia","start":{"date-parts":[[2026,1,31]]},"end":{"date-parts":[[2026,2,4]]}},"container-title":["2026 IEEE International Symposium on High Performance Computer Architecture (HPCA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11408404\/11408433\/11408597.pdf?arnumber=11408597","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T06:36:18Z","timestamp":1772692578000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11408597\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,31]]},"references-count":71,"URL":"https:\/\/doi.org\/10.1109\/hpca68181.2026.11408597","relation":{},"subject":[],"published":{"date-parts":[[2026,1,31]]}}}