{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T12:48:11Z","timestamp":1782996491225,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":118,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["2450448"],"award-info":[{"award-number":["2450448"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["2312688"],"award-info":[{"award-number":["2312688"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,6]]},"DOI":"10.1145\/3797905.3800514","type":"proceedings-article","created":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T11:50:37Z","timestamp":1782993037000},"page":"145-160","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Wattchmen: Watching the Wattchers \u2013 High Fidelity, Flexible GPU Energy Modeling"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8190-628X","authenticated-orcid":false,"given":"Brandon","family":"Tran","sequence":"first","affiliation":[{"name":"Computer Sciences, University of Wisconsin-Madison, Madison, Wisconsin, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8698-460X","authenticated-orcid":false,"given":"Matthias","family":"Maiterth","sequence":"additional","affiliation":[{"name":"NVIDIA, Santa Clara, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7207-7814","authenticated-orcid":false,"given":"Woong","family":"Shin","sequence":"additional","affiliation":[{"name":"Oak Ridge National Laboratory, Oak Ridge, Tennessee, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0189-7895","authenticated-orcid":false,"given":"Matthew D.","family":"Sinclair","sequence":"additional","affiliation":[{"name":"Computer Sciences, University of Wisconsin\u2013Madison, Madison, Wisconsin, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9575-7935","authenticated-orcid":false,"given":"Shivaram","family":"Venkataraman","sequence":"additional","affiliation":[{"name":"Computer Sciences, University of Wisconsin-Madison, Madison, Wisconsin, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,5]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","DOI":"10.1109\/HiPC.2017.00020"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2014.18"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","unstructured":"Dong\u00a0H. Ahn Ned Bass Albert Chu Jim Garlick Mark Grondona Stephen Herbein Helgi\u00a0I. Ing\u00f3lfsson Joseph Koning Tapasya Patki Thomas\u00a0R.W. Scogland Becky Springmeyer and Michela Taufer. 2020. Flux: Overcoming scheduling challenges for exascale workflows. Future Generation Computer Systems 110 (2020) 202\u2013213. 10.1016\/j.future.2020.04.006","DOI":"10.1016\/j.future.2020.04.006"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","unstructured":"Gargi Alavani Jineet Desai Snehanshu Saha and Santonu Sarkar. 2023. Program Analysis and Machine Learning\u2013based Approach to Predict Power Consumption of CUDA Kernel. ACM Trans. Model. Perform. Eval. Comput. Syst. 8 4 Article 10 (July 2023) 24\u00a0pages. 10.1145\/3603533","DOI":"10.1145\/3603533"},{"key":"e_1_3_3_1_6_2","unstructured":"AMD. 2020. AMD MxGPU and VMware. https:\/\/drivers.amd.com\/relnotes\/amd_mxgpu_deploymentguide_vmware.pdf."},{"key":"e_1_3_3_1_7_2","volume-title":"AMD Instinct\u2122 MI250 microarchitecture","year":"2025","unstructured":"AMD. 2025. AMD Instinct\u2122 MI250 microarchitecture. AMD. Retrieved April 14, 2025 from https:\/\/rocm.docs.amd.com\/en\/latest\/conceptual\/gpu-arch\/mi250.html"},{"key":"e_1_3_3_1_8_2","unstructured":"AMD. 2025. ROCm Data Center (RDC) tool documentation \u2014 ROCm Data Center Documentation. https:\/\/rocm.docs.amd.com\/projects\/rdc\/en\/latest\/"},{"key":"e_1_3_3_1_9_2","unstructured":"AMD. 2025. ROCm System Management Interface (ROCm SMI) library \u2014 ROCm SMI LIB 7.6.0 Documentation. https:\/\/rocm.docs.amd.com\/projects\/rocm_smi_lib\/en\/latest\/"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.2172\/1822199"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.1145\/3387902.3392613"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607089"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/3339186.3339215"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"publisher","DOI":"10.1145\/3203217.3205863"},{"key":"e_1_3_3_1_15_2","unstructured":"Keren Bergman Shekhar Borkar Dan Campbell William Carlson William Dally Monty Denneau Paul Franzon William Harrod Jon Hiller Sherman Karp Stephen Keckler Dean Klein Robert Lucas Mark Richards Al Scarpelli Steven Scott Allan Snavely Thomas Sterling R.\u00a0Stanley Williams Katherine Yelick Keren Bergman Shekhar Borkar Dan Campbell William Carlson William Dally Monty Denneau Paul Franzon William Harrod Jon Hiller Stephen Keckler Dean Klein Peter Kogge R.\u00a0Stanley Williams and Katherine Yelick. 2008. ExaScale Computing Study: Technology Challenges in Achieving Exascale Systems."},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","unstructured":"Andrea Borghesi Alessio Burrello and Andrea Bartolini. 2023. ExaMon-X: A Predictive Maintenance Framework for Automatic Monitoring in Industrial IoT Systems. IEEE Internet of Things Journal 10 4 (Feb. 2023) 2995\u20133005. 10.1109\/JIOT.2021.3125885","DOI":"10.1109\/JIOT.2021.3125885"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/339647.339657"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.1145\/344166.344576"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2013.6704684"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2010.5650274"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2014.54"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2013.77"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","unstructured":"Jee\u00a0Whan Choi and Richard\u00a0W. Vuduc. 2013. How much (execution) time and energy does my algorithm cost? XRDS: Crossroads The ACM Magazine for Students 19 3 (March 2013) 49\u201351. 10.1145\/2425676.2425691","DOI":"10.1145\/2425676.2425691"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","unstructured":"Jack Choquette. 2023. NVIDIA Hopper H100 GPU: Scaling Performance. IEEE Micro 43 03 (May 2023) 9\u201317. 10.1109\/MM.2023.3256796","DOI":"10.1109\/MM.2023.3256796"},{"key":"e_1_3_3_1_26_2","unstructured":"SuiteSparse\u00a0Matrix Collection. 2001. ATandT\/pre2 AT&T harmonic balance method large example. https:\/\/sparse.tamu.edu\/ATandT\/pre2."},{"key":"e_1_3_3_1_27_2","unstructured":"Cray. 2025. Cray Performance Measurement and Analysis Tools User Guide 6.4.5 S-2376. https:\/\/support.hpe.com\/hpesc\/public\/docDisplay?docId=a00113915en_us"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO61859.2024.00058"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"publisher","DOI":"10.1145\/1840845.1840883"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/ASAP61560.2024.00038"},{"key":"e_1_3_3_1_31_2","unstructured":"DMTF. 2025. REDFISH | DMTF. https:\/\/www.dmtf.org\/standards\/redfish"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"publisher","unstructured":"Jack Dongarra Steven Gottlieb and William T.\u00a0C. Kramer. 2019. Race to Exascale. Computing in Science & Engineering 21 1 (2019) 4\u20135. 10.1109\/MCSE.2018.2882574","DOI":"10.1109\/MCSE.2018.2882574"},{"key":"e_1_3_3_1_33_2","volume-title":"Top500","author":"Dongarra Jack\u00a0J.","year":"2024","unstructured":"Jack\u00a0J. Dongarra, Hans\u00a0W. Meuer, and Erich Strohmaier. 2024. Top500. TOP500. Retrieved Nov 18, 2024 from https:\/\/www.top500.org\/"},{"key":"e_1_3_3_1_34_2","first-page":"1","volume-title":"2019 USENIX Annual Technical Conference (USENIX ATC 19)","author":"Duplyakin Dmitry","year":"2019","unstructured":"Dmitry Duplyakin, Robert Ricci, Aleksander Maricq, Gary Wong, Jonathon Duerig, Eric Eide, Leigh Stoller, Mike Hibler, David Johnson, Kirk Webb, Aditya Akella, Kuangching Wang, Glenn Ricart, Larry Landweber, Chip Elliott, Michael Zink, Emmanuel Cecchet, Snigdhaswin Kar, and Prabodh Mishra. 2019. The Design and Operation of CloudLab. In 2019 USENIX Annual Technical Conference (USENIX ATC 19). USENIX Association, Renton, WA, 1\u201314. https:\/\/www.usenix.org\/conference\/atc19\/presentation\/duplyakin"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-58667-0_21"},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"publisher","DOI":"10.1145\/2000064.2000108"},{"key":"e_1_3_3_1_37_2","doi-asserted-by":"publisher","unstructured":"M. Etinski Julita Corbalan Jes\u00fas Labarta and Mateo Valero. 2012. Understanding the future of energy-performance trade-off via DVFS in HPC environments. J. Parallel and Distrib. Comput. 72 (04 2012) 579\u2013590. 10.1016\/j.jpdc.2012.01.006","DOI":"10.1016\/j.jpdc.2012.01.006"},{"key":"e_1_3_3_1_38_2","doi-asserted-by":"publisher","DOI":"10.1145\/3337821.3337833"},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"publisher","DOI":"10.1109\/CCGRID64434.2025.00035"},{"key":"e_1_3_3_1_40_2","doi-asserted-by":"publisher","unstructured":"Millad Ghane Jeff Larkin Larry Shi Sunita Chandrasekaran and Margaret\u00a0S. Cheung. 2018. Power and Energy-efficiency Roofline Model for GPUs. 10.48550\/arXiv.1809.09206arXiv:https:\/\/arXiv.org\/abs\/1809.09206 [cs].","DOI":"10.48550\/arXiv.1809.09206"},{"key":"e_1_3_3_1_41_2","doi-asserted-by":"publisher","unstructured":"William\u00a0F. Godoy Steven\u00a0E. Hahn Michael\u00a0M. Walsh Philip\u00a0W. Fackler Jaron\u00a0T. Krogel Peter\u00a0W. Doak Paul\u00a0R.C. Kent Alfredo\u00a0A. Correa Ye Luo and Mark Dewing. 2025. Software Stewardship and Advancement of a High-performance Computing Scientific Application: QMCPACK. Future Generation Computer Systems 163 (2025) 107502. 10.1016\/j.future.2024.107502","DOI":"10.1016\/j.future.2024.107502"},{"key":"e_1_3_3_1_42_2","doi-asserted-by":"publisher","DOI":"10.1145\/3624062.3624274"},{"key":"e_1_3_3_1_43_2","unstructured":"R.\u00a0E. Grant M. Levenhagen S.\u00a0L. Olivier D. DeBonis K.\u00a0T. Pedretti and J.\u00a0H. Laros\u00a0III. 2025. Power API. https:\/\/pwrapi.github.io\/"},{"key":"e_1_3_3_1_44_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2018.00072"},{"key":"e_1_3_3_1_45_2","doi-asserted-by":"publisher","unstructured":"Jo\u00e3o Guerreiro Aleksandar Ilic Nuno Roma and Pedro Tom\u00e1s. 2019. Modeling and Decoupling the GPU Power Consumption for Cross-Domain DVFS. IEEE Transactions on Parallel and Distributed Systems 30 11 (2019) 2494\u20132506. 10.1109\/TPDS.2019.2917181","DOI":"10.1109\/TPDS.2019.2917181"},{"key":"e_1_3_3_1_46_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2018.00058"},{"key":"e_1_3_3_1_47_2","doi-asserted-by":"publisher","DOI":"10.1145\/3712285.3759768"},{"key":"e_1_3_3_1_48_2","unstructured":"HPE. 2025. PM Counters | HPE Cray Supercomputing User Services Software Administration Guide for HPE Performance Cluster Manager (1.2.0) (S-8064). https:\/\/support.hpe.com\/hpesc\/public\/docDisplay?docId=dp00005587en_us&page=operations-hpcm\/os\/PM_Counters.html&docLocale=en_US"},{"key":"e_1_3_3_1_49_2","unstructured":"Rodrigo Huerta Mojtaba\u00a0Abaie Shoushtary Jos\u00e9-Lorenzo Cruz and Antonio Gonz\u00e1lez. 2025. Analyzing Modern NVIDIA GPU cores. arxiv:https:\/\/arXiv.org\/abs\/2503.20481\u00a0[cs.AR] https:\/\/arxiv.org\/abs\/2503.20481"},{"key":"e_1_3_3_1_50_2","volume-title":"4th gem5 Users\u2019 Workshop","author":"Jamieson Charles","year":"2022","unstructured":"Charles Jamieson, Anushka Chandrashekar, Ian McDougall, and Matthew\u00a0D. Sinclair. 2022. GAP: gem5 GPU Accuracy Profiler. In 4th gem5 Users\u2019 Workshop. Association for Computing Machinery, New York, NY, USA, 2\u00a0pages."},{"key":"e_1_3_3_1_51_2","doi-asserted-by":"publisher","DOI":"10.1145\/3466752.3480063"},{"key":"e_1_3_3_1_52_2","unstructured":"Paul Kent. 2025. Exascale Computing Project QMCPACK. https:\/\/www.exascaleproject.org\/research-project\/qmcpack\/"},{"key":"e_1_3_3_1_53_2","doi-asserted-by":"publisher","unstructured":"P.\u00a0R.\u00a0C. Kent Abdulgani Annaberdiyev Anouar Benali M.\u00a0Chandler Bennett Edgar\u00a0Josu\u00e9 Landinez\u00a0Borda Peter Doak Hongxia Hao Kenneth\u00a0D. Jordan Jaron\u00a0T. Krogel Ilkka Kyl\u00e4np\u00e4\u00e4 Joonho Lee Ye Luo Fionn\u00a0D. Malone Cody\u00a0A. Melton Lubos Mitas Miguel\u00a0A. Morales Eric Neuscamman Fernando\u00a0A. Reboredo Brenda Rubenstein Kayahan Saritas Shiv Upadhyay Guangming Wang Shuai Zhang and Luning Zhao. 2020. QMCPACK: Advances in the development efficiency and application of auxiliary field and real-space variational and diffusion quantum Monte Carlo. The Journal of Chemical Physics 152 17 (05 2020) 174105. arXiv:https:\/\/pubs.aip.org\/aip\/jcp\/article-pdf\/doi\/10.1063\/5.0004860\/16740875\/174105_1_online.pdf10.1063\/5.0004860","DOI":"10.1063\/5.0004860"},{"key":"e_1_3_3_1_54_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA45697.2020.00047"},{"key":"e_1_3_3_1_55_2","doi-asserted-by":"publisher","unstructured":"Jeongnim Kim Andrew\u00a0D Baczewski Todd\u00a0D Beaudet Anouar Benali M\u00a0Chandler Bennett Mark\u00a0A Berrill Nick\u00a0S Blunt Edgar Josu\u00e9\u00a0Landinez Borda Michele Casula David\u00a0M Ceperley Simone Chiesa Bryan\u00a0K Clark Raymond\u00a0C Clay Kris\u00a0T Delaney Mark Dewing Kenneth\u00a0P Esler Hongxia Hao Olle Heinonen Paul R\u00a0C Kent Jaron\u00a0T Krogel Ilkka Kyl\u00e4np\u00e4\u00e4 Ying\u00a0Wai Li M\u00a0Graham Lopez Ye Luo Fionn\u00a0D Malone Richard\u00a0M Martin Amrita Mathuriya Jeremy McMinis Cody\u00a0A Melton Lubos Mitas Miguel\u00a0A Morales Eric Neuscamman William\u00a0D Parker Sergio\u00a0D Pineda\u00a0Flores Nichols\u00a0A Romero Brenda\u00a0M Rubenstein Jacqueline A\u00a0R Shea Hyeondeok Shin Luke Shulenburger Andreas\u00a0F Tillack Joshua\u00a0P Townsend Norm\u00a0M Tubman Brett Van Der\u00a0Goetz Jordan\u00a0E Vincent D\u00a0ChangMo Yang Yubo Yang Shuai Zhang and Luning Zhao. 2018. QMCPACK: An Open Source ab initio Quantum Monte Carlo Package for the Electronic Structure of Atoms Molecules and Solids. Journal of Physics: Condensed Matter 30 19 (4 2018) 195901. 10.1088\/1361-648X\/aab9c3","DOI":"10.1088\/1361-648X\/aab9c3"},{"key":"e_1_3_3_1_56_2","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER49012.2020.00069"},{"key":"e_1_3_3_1_57_2","doi-asserted-by":"publisher","unstructured":"Douglas Kothe Stephen Lee and Irene Qualters. 2019. Exascale Computing in the United States. Computing in Science & Engineering 21 1 (2019) 17\u201329. 10.1109\/MCSE.2018.2875366","DOI":"10.1109\/MCSE.2018.2875366"},{"key":"e_1_3_3_1_58_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-43222-5_11"},{"key":"e_1_3_3_1_59_2","unstructured":"Lawrence Livermore National Labs. 2020. CORAL-2 Benchmarks. https:\/\/asc.llnl.gov\/coral-2-benchmarks."},{"key":"e_1_3_3_1_60_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2014.6844477"},{"key":"e_1_3_3_1_61_2","doi-asserted-by":"publisher","DOI":"10.1145\/2485922.2485964"},{"key":"e_1_3_3_1_62_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2019.00028"},{"key":"e_1_3_3_1_63_2","doi-asserted-by":"publisher","unstructured":"Sheng Li Jung\u00a0Ho Ahn Richard\u00a0D. Strong Jay\u00a0B. Brockman Dean\u00a0M. Tullsen and Norman\u00a0P. Jouppi. 2013. The McPAT Framework for Multicore and Manycore Architectures: Simultaneously Modeling Power Area and Timing. ACM Transactions on Architecture & Code Optimization 10 1 Article 5 (April 2013) 29\u00a0pages. 10.1145\/2445572.2445577","DOI":"10.1145\/2445572.2445577"},{"key":"e_1_3_3_1_64_2","doi-asserted-by":"publisher","unstructured":"V. Ljungdahl M. Jradi and C. Veje. 2022. A decision support model for waste heat recovery systems design in Data Center and High-Performance Computing clusters utilizing liquid cooling and Phase Change Materials. Applied Thermal Engineering 201 (2022) 117671. 10.1016\/j.applthermaleng.2021.117671","DOI":"10.1016\/j.applthermaleng.2021.117671"},{"key":"e_1_3_3_1_65_2","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589349"},{"key":"e_1_3_3_1_66_2","unstructured":"Jason Lowe-Power Abdul\u00a0Mutaal Ahmad Ayaz Akram Mohammad Alian Rico Amslinger Matteo Andreozzi Adri\u00e0 Armejach Nils Asmussen Srikant Bharadwaj Gabe Black Gedare Bloom Bobby\u00a0R. Bruce Daniel\u00a0Rodrigues Carvalho Jeronimo Castrillon Lizhong Chen Nicolas Derumigny Stephan Diestelhorst Wendy Elsasser Marjan Fariborz Amin Farmahini-Farahani Pouya Fotouhi Ryan Gambord Jayneel Gandhi Dibakar Gope Thomas Grass Bagus Hanindhito Andreas Hansson Swapnil Haria Austin Harris Timothy Hayes Adrian Herrera Matthew Horsnell Syed Ali\u00a0Raza Jafri Radhika Jagtap Hanhwi Jang Reiley Jeyapaul Timothy\u00a0M. Jones Matthias Jung Subash Kannoth Hamidreza Khaleghzadeh Yuetsu Kodama Tushar Krishna Tommaso Marinelli Christian Menard Andrea Mondelli Tiago M\u00fcck Omar Naji Krishnendra Nathella Hoa Nguyen Nikos Nikoleris Lena\u00a0E. Olson Marc Orr Binh Pham Pablo Prieto Trivikram Reddy Alec Roelke Mahyar Samani Andreas Sandberg Javier Setoain Boris Shingarov Matthew\u00a0D. Sinclair Tuan Ta Rahul Thakur Giacomo Travaglini Michael Upton Nilay Vaish Ilias Vougioukas Zhengrong Wang Norbert Wehn Christian Weis David\u00a0A. Wood Hongil Yoon and \u00c9der F.\u00a0Zulian. 2020. The gem5 Simulator: Version 20.0+. arxiv:https:\/\/arXiv.org\/abs\/2007.03152\u00a0[cs.AR]"},{"key":"e_1_3_3_1_67_2","doi-asserted-by":"publisher","DOI":"10.1145\/3731599.3767559"},{"key":"e_1_3_3_1_68_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2018.00111"},{"key":"e_1_3_3_1_69_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCS.2017.11"},{"key":"e_1_3_3_1_70_2","unstructured":"Moiralex. 2024. Issues with Accelwattch Quadratic Programming Optimization. https:\/\/github.com\/accel-sim\/accel-sim-framework\/discussions\/314."},{"key":"e_1_3_3_1_71_2","doi-asserted-by":"publisher","unstructured":"Trevor Mudge. 2001. Power: a First-class Architectural Design Constraint. Computer 34 4 (2001) 52\u201358. 10.1109\/2.917539","DOI":"10.1109\/2.917539"},{"key":"e_1_3_3_1_72_2","doi-asserted-by":"crossref","unstructured":"Naveen Muralimanohar Rajeev Balasubramonian and Norman\u00a0P Jouppi. 2009. CACTI 6.0: A tool to model large caches. HP Laboratories 27 (2009) 28.","DOI":"10.1109\/MM.2008.2"},{"key":"e_1_3_3_1_73_2","first-page":"16","volume-title":"IEEE International Solid-State Circuits Conference","author":"Naffziger S.","year":"2005","unstructured":"S. Naffziger, J. Warnock, and H. Knapp. 2005. When Processors Hit the Power Wall (or \u201cWhen the CPU hits the fan\u201d). In IEEE International Solid-State Circuits Conference. IEEE, Piscataway, NJ, USA, 16\u201317."},{"key":"e_1_3_3_1_74_2","unstructured":"Sharan Narang. 2016. DeepBench. https:\/\/github.com\/baidu-research\/DeepBench."},{"key":"e_1_3_3_1_75_2","unstructured":"Sharan Narang and Greg Diamos. 2017. An update to DeepBench with a focus on deep learning inference. https:\/\/svail.github.io\/DeepBench-update\/."},{"key":"e_1_3_3_1_76_2","doi-asserted-by":"publisher","DOI":"10.1145\/3369583.3392674"},{"key":"e_1_3_3_1_77_2","unstructured":"NVIDIA. 2025. DCGM Diagnostics \u2014 NVIDIA DCGM Documentation latest documentation. https:\/\/docs.nvidia.com\/datacenter\/dcgm\/latest\/user-guide\/dcgm-diagnostics.html"},{"key":"e_1_3_3_1_78_2","unstructured":"NVIDIA. 2025. Nsight Compute Documentation \u2014 NsightCompute 12.8 documentation. https:\/\/docs.nvidia.com\/nsight-compute\/index.html"},{"key":"e_1_3_3_1_79_2","unstructured":"NVIDIA. 2025. NVIDIA Management Library (NVML). https:\/\/developer.nvidia.com\/management-library-nvml"},{"key":"e_1_3_3_1_80_2","volume-title":"NVIDIA Multi-Instance GPU","year":"2025","unstructured":"NVIDIA. 2025. NVIDIA Multi-Instance GPU. NVIDIA. Retrieved April 14, 2025 from https:\/\/www.nvidia.com\/en-us\/technologies\/multi-instance-gpu\/"},{"key":"e_1_3_3_1_81_2","unstructured":"NVIDIA. 2025. Parallel Thread Execution ISA Version 8.7. https:\/\/docs.nvidia.com\/cuda\/parallel-thread-execution\/."},{"key":"e_1_3_3_1_82_2","unstructured":"NVIDIA. 2025. User Guide \u2014 nsight-systems 2025.2 documentation. https:\/\/docs.nvidia.com\/nsight-systems\/UserGuide\/index.html"},{"key":"e_1_3_3_1_83_2","unstructured":"Oak Ridge National Laboratories. 2025. OLCF-6 Benchmarks. https:\/\/www.olcf.ornl.gov\/draft-olcf-6-technical-requirements\/benchmarks\/"},{"key":"e_1_3_3_1_84_2","doi-asserted-by":"publisher","DOI":"10.1145\/3650200.3656615"},{"key":"e_1_3_3_1_85_2","unstructured":"ORNL. 2022. Oak Ridge National Laboratory. https:\/\/www.olcf.ornl.gov\/summit\/."},{"key":"e_1_3_3_1_86_2","unstructured":"Lawrence Page Sergey Brin Rajeev Motwani and Terry Winograd. 1999. The PageRank Citation Ranking: Bringing Order to the Web.http:\/\/ilpubs.stanford.edu:8090\/422\/ Previous number = SIDL-WP-1999-0120."},{"key":"e_1_3_3_1_87_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651329"},{"key":"e_1_3_3_1_88_2","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC55918.2022.00033"},{"key":"e_1_3_3_1_89_2","doi-asserted-by":"publisher","DOI":"10.1109\/WORKS49585.2019.00009"},{"key":"e_1_3_3_1_90_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2012.116"},{"key":"e_1_3_3_1_91_2","doi-asserted-by":"publisher","DOI":"10.1109\/PMBS56514.2022.00010"},{"key":"e_1_3_3_1_92_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-19328-6_1"},{"key":"e_1_3_3_1_93_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00087"},{"key":"e_1_3_3_1_94_2","volume-title":"2024 United States Data Center Energy Usage Report","author":"Shehabi Arman","year":"2024","unstructured":"Arman Shehabi, Alex Newkirk, Sarah\u00a0J Smith, Alex Hubbard, Nuoa Lei, Md\u00a0Abu\u00a0Bakar Siddik, Billie Holecek, Jonathan Koomey, Eric Masanet, and Dale Sartor. 2024. 2024 United States Data Center Energy Usage Report. Technical Report LBNL-2001637. Lawrence Berkeley National Laboratory."},{"key":"e_1_3_3_1_95_2","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476188"},{"key":"e_1_3_3_1_96_2","doi-asserted-by":"publisher","DOI":"10.1109\/SCW63240.2024.00225"},{"key":"e_1_3_3_1_97_2","unstructured":"Mohammad Shoeybi Mostofa Patwary Raul Puri Patrick LeGresley Jared Casper and Bryan Catanzaro. 2019. Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. arxiv:https:\/\/arXiv.org\/abs\/1909.08053\u00a0[cs.CL]"},{"key":"e_1_3_3_1_98_2","series-title":"(OSCAR)","volume-title":"3rd Open-Source Computer Architecture Research Workshop","author":"Smith Alex","year":"2024","unstructured":"Alex Smith, Bobby Bruce, Jason Lowe-Power, and M.\u00a0D. Sinclair. 2024. Designing Generalizable Power Models For Open-Source Architecture Simulators. In 3rd Open-Source Computer Architecture Research Workshop(OSCAR). Association for Computing Machinery, New York, NY, USA, 4\u00a0pages."},{"key":"e_1_3_3_1_99_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00068"},{"key":"e_1_3_3_1_100_2","volume-title":"6th gem5 Users Workshop","author":"Smith Alex","year":"2025","unstructured":"Alex Smith and M.\u00a0D. Sinclair. 2025. Implementing Support for Extensible Power Modeling in gem5. In 6th gem5 Users Workshop. Association for Computing Machinery, New York, NY, USA, 2\u00a0pages."},{"key":"e_1_3_3_1_101_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41406.2024.00030"},{"key":"e_1_3_3_1_102_2","doi-asserted-by":"publisher","DOI":"10.1145\/3676641.3716025"},{"key":"e_1_3_3_1_103_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00102"},{"key":"e_1_3_3_1_104_2","unstructured":"Yifan Sun Nicolas\u00a0Bohm Agostini Shi Dong and David Kaeli. 2020. Summarizing CPU and GPU Design Trends with Product Data. arxiv:https:\/\/arXiv.org\/abs\/1911.11313\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/1911.11313"},{"key":"e_1_3_3_1_105_2","unstructured":"TACC. 2024. Texas Advanced Computing Center. https:\/\/www.tacc.utexas.edu\/."},{"key":"e_1_3_3_1_106_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC43674.2020.9286239"},{"key":"e_1_3_3_1_107_2","doi-asserted-by":"crossref","unstructured":"Torsten Wilde Axel Auweter and Hayk Shoukourian. 2014. The 4 Pillar Framework for energy efficient HPC data centers. SICS Software-Intensive Cyber-Physical Systems 29 (2014) 241\u2013251. https:\/\/link.springer.com\/article\/10.1007\/s00450-013-0244-6","DOI":"10.1007\/s00450-013-0244-6"},{"key":"e_1_3_3_1_108_2","series-title":"(ASHES)","first-page":"1","volume-title":"Sixteenth International Workshop on Accelerators and Hybrid Emerging Systems","author":"Tran Brandon","year":"2026","unstructured":"Brandon Tran, Matthias Maiterth, Woong Shin, Matthew\u00a0D. Sinclair, and Shivaram Venkataraman. 2026. ET: Bridging the Gap on Energy Telemetry for Multi-GPU Communication Collectives. In Sixteenth International Workshop on Accelerators and Hybrid Emerging Systems(ASHES). IEEE Computer Society, Los Alamitos, CA, USA, 1\u20134."},{"key":"e_1_3_3_1_109_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00092"},{"key":"e_1_3_3_1_110_2","unstructured":"Variorum. 2025. Variorum \u2014 Variorum 0.7.0 documentation. https:\/\/variorum.readthedocs.io\/en\/latest\/index.html#"},{"key":"e_1_3_3_1_111_2","volume-title":"OpenBMC Event Subscription Protocol","author":"Williams Patrick","year":"2015","unstructured":"Patrick Williams. 2015. OpenBMC Event Subscription Protocol. OpenBMC. Retrieved April 2, 2021 from https:\/\/github.com\/openbmc\/docs\/blob\/master\/rest-api.md"},{"key":"e_1_3_3_1_112_2","doi-asserted-by":"publisher","DOI":"10.1145\/3624062.3624262"},{"key":"e_1_3_3_1_113_2","doi-asserted-by":"crossref","unstructured":"S.J.E. Wilton and N.P. Jouppi. 1996. CACTI: An Enhanced Cache Access and Cycle Time Model. Solid-State Circuits IEEE Journal of 31 5 (5 1996) 677\u2013688.","DOI":"10.1109\/4.509850"},{"key":"e_1_3_3_1_114_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2015.7056063"},{"key":"e_1_3_3_1_115_2","unstructured":"Shawn Xiaolong. 2024. SASS SIM mode has no power output But PTX SIM mode has. https:\/\/github.com\/accel-sim\/accel-sim-framework\/issues\/347."},{"key":"e_1_3_3_1_116_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41406.2024.00028"},{"key":"e_1_3_3_1_117_2","series-title":"(NSDI)","first-page":"119","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation","author":"You Jie","year":"2023","unstructured":"Jie You, Jae-Won Chung, and Mosharaf Chowdhury. 2023. Zeus: Understanding and Optimizing GPU Energy Consumption of DNN Training. In 20th USENIX Symposium on Networked Systems Design and Implementation(NSDI). USENIX Association, Boston, MA, 119\u2013139. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/you"},{"key":"e_1_3_3_1_118_2","doi-asserted-by":"publisher","DOI":"10.1145\/3624062.3624275"},{"key":"e_1_3_3_1_119_2","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC47752.2019.9041972"}],"event":{"name":"ICS '26: 2026 International Conference on Supercomputing","location":"Belfast United Kingdom","acronym":"ICS '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 40th ACM International Conference on Supercomputing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/abs\/10.1145\/3797905.3800514","content-type":"text\/html","content-version":"vor","intended-application":"syndication"}],"deposited":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T12:31:44Z","timestamp":1782995504000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3797905.3800514"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":118,"alternative-id":["10.1145\/3797905.3800514","10.1145\/3797905"],"URL":"https:\/\/doi.org\/10.1145\/3797905.3800514","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}