{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T13:44:32Z","timestamp":1782999872978,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":75,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["CNS-2336886"],"award-info":[{"award-number":["CNS-2336886"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,6]]},"DOI":"10.1145\/3797905.3800546","type":"proceedings-article","created":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T11:50:37Z","timestamp":1782993037000},"page":"188-200","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["CATS: Correlation-aware Task Scheduling for GPU Power Optimization in AI Data Centers"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5848-5667","authenticated-orcid":false,"given":"Srinivasan","family":"Subramaniyan","sequence":"first","affiliation":[{"name":"Electrical and Computer Engineering, The Ohio State University, Columbus, Ohio, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9633-1418","authenticated-orcid":false,"given":"Xiaorui","family":"Wang","sequence":"additional","affiliation":[{"name":"Electrical and Computer Engineering, The Ohio State University, Columbus, Ohio, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,5]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"[n. d.]. Tech\u2019s splurge on AI chips has companies in \u2019arms race\u2019 that\u2019s forcing more spending. https:\/\/www.cnbc.com\/2024\/07\/25\/techs-splurge-on-ai-chips-has-meta-alphabet-tesla-in-arms-race.html."},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid54584.2022.00079"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1145\/3387902.3392613"},{"key":"e_1_3_3_1_5_2","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation (OSDI)","author":"Bai Zhihao","year":"2020","unstructured":"Zhihao Bai, Zhen Zhang, Yibo Zhu, and Xin Jin. 2020. PipeSwitch: Fast pipelined context switching for deep learning applications. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI)."},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2019.8737460"},{"key":"e_1_3_3_1_7_2","volume-title":"Workshop on ML Systems, NIPS","author":"Boag Scott","year":"2017","unstructured":"Scott Boag, Parijat Dube, Benjamin Herta, Waldemar Hummer, Vatche Ishakian, K Jayaram, Michael Kalantar, Vinod Muthusamy, Priya Nagpurkar, and Florian Rosenberg. 2017. Scalable multi-framework multi-tenant lifecycle management of deep learning training jobs. In Workshop on ML Systems, NIPS."},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"crossref","unstructured":"Robert\u00a0A Bridges Neena Imam and Tiffany\u00a0M Mintz. 2016. Understanding GPU power: A survey of profiling modeling and simulation methods. ACM Computing Surveys (CSUR) 49 3 (2016).","DOI":"10.1145\/2962131"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS60910.2024.00051"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS54860.2022.00039"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607060"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.23919\/DATE.2018.8341972"},{"key":"e_1_3_3_1_13_2","volume-title":"2022 USENIX Annual Technical Conference (USENIX ATC)","author":"Choi Seungbeom","year":"2022","unstructured":"Seungbeom Choi, Sunho Lee, Yeonjae Kim, Jongse Park, Youngjin Kwon, and Jaehyuk Huh. 2022. Serving heterogeneous machine learning models on Multi-GPU servers with Spatio-Temporal sharing. In 2022 USENIX Annual Technical Conference (USENIX ATC)."},{"key":"e_1_3_3_1_14_2","volume-title":"NVIDIA System Management Interface","author":"Corporation NVIDIA","year":"2023","unstructured":"NVIDIA Corporation. 2023. NVIDIA System Management Interface."},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.23919\/DATE58400.2024.10546769"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421284"},{"key":"e_1_3_3_1_17_2","unstructured":"Paul Elvinger Foteini Strati Natalie\u00a0Enright Jerger and Ana Klimovic. 2025. Measuring GPU utilization one level deeper. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.16909 (2025)."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.1109\/FiCloud49777.2021.00063"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3472883.3486978"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/3368089.3417050"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Guin Gilman and Robert\u00a0J Walls. 2022. Characterizing concurrency mechanisms for NVIDIA GPUs under deep learning workloads. ACM SIGMETRICS Performance Evaluation Review 49 3 (2022).","DOI":"10.1145\/3529113.3529124"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"crossref","unstructured":"Robert Grandl Ganesh Ananthanarayanan Srikanth Kandula Sriram Rao and Aditya Akella. 2014. Multi-resource packing for cluster schedulers. ACM SIGCOMM Computer Communication Review 44 4 (2014).","DOI":"10.1145\/2740070.2626334"},{"key":"e_1_3_3_1_23_2","unstructured":"Diandian Gu Xintong Xie Gang Huang Xin Jin and Xuanzhe Liu. 2023. Energy-efficient GPU clusters scheduling for deep learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.06381 (2023)."},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM41043.2020.9155445"},{"key":"e_1_3_3_1_25_2","volume-title":"NSDI","author":"Heller Brandon","year":"2010","unstructured":"Brandon Heller, Srinivasan Seetharaman, Priya Mahadevan, Yiannis Yiakoumis, Puneet Sharma, Sujata Banerjee, and Nick McKeown. 2010. Elastictree: Saving energy in data center networks.. In NSDI."},{"key":"e_1_3_3_1_26_2","unstructured":"Andrew\u00a0G Howard Menglong Zhu Bo Chen Dmitry Kalenichenko Weijun Wang Tobias Weyand Marco Andreetto and Hartwig Adam. 2017. Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1704.04861 (2017)."},{"key":"e_1_3_3_1_27_2","unstructured":"Ching-Hsien Hsu Kenn\u00a0D Slagter Shih-Chang Chen and Yeh-Ching Chung. 2014. Optimizing energy consumption with task consolidation in clouds. Elsevier Information Sciences (2014)."},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476223"},{"key":"e_1_3_3_1_29_2","unstructured":"Forrest\u00a0N Iandola Song Han Matthew\u00a0W Moskewicz Khalid Ashraf William\u00a0J Dally and Kurt Keutzer. 2016. SqueezeNet: AlexNet-level accuracy with 50x fewer parameters and< 0.5 MB model size. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1602.07360 (2016)."},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.1145\/3767295.3769333"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00049"},{"key":"e_1_3_3_1_32_2","volume-title":"USENIX Annual Technical Conference (USENIX ATC)","author":"Jeon Myeongjae","year":"2019","unstructured":"Myeongjae Jeon, Shivaram Venkataraman, Amar Phanishayee, Junjie Qian, Wencong Xiao, and Fan Yang. 2019. Analysis of Large-Scale Multi-Tenant GPU clusters for DNN training workloads. In USENIX Annual Technical Conference (USENIX ATC)."},{"key":"e_1_3_3_1_33_2","unstructured":"Zhihao Jia Matei Zaharia and Alex Aiken. 2019. Beyond data and model parallelism for deep neural networks. Proceedings of Machine Learning and Systems (2019)."},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"crossref","unstructured":"Andreas\u00a0Kosmas Kakolyris Dimosthenis Masouros Sotirios Xydis and Dimitrios Soudris. 2024. Slo-aware gpu dvfs for energy-efficient llm inference serving. IEEE Computer Architecture Letters (2024).","DOI":"10.1109\/LCA.2024.3406038"},{"key":"e_1_3_3_1_35_2","unstructured":"Beth Kindig. 2024. AI Power Consumption: Rapidly Becoming Mission-Critical. https:\/\/www.forbes.com\/sites\/bethkindig\/2024\/06\/20\/ai-power-consumption-rapidly-becoming-mission-critical\/."},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41406.2024.00048"},{"key":"e_1_3_3_1_37_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICAC.2007.35"},{"key":"e_1_3_3_1_38_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607034"},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"publisher","DOI":"10.1145\/2640087.2644155"},{"key":"e_1_3_3_1_40_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2011.129"},{"key":"e_1_3_3_1_41_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_3_1_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2012.31"},{"key":"e_1_3_3_1_43_2","doi-asserted-by":"publisher","DOI":"10.1145\/3754598.3754670"},{"key":"e_1_3_3_1_44_2","doi-asserted-by":"crossref","unstructured":"David Meisner Brian\u00a0T Gold and Thomas\u00a0F Wenisch. 2009. Powernap: eliminating server idle power. ACM SIGARCH Computer Architecture News (2009).","DOI":"10.1145\/2528521.1508269"},{"key":"e_1_3_3_1_45_2","unstructured":"Stuart Mitchell Michael OSullivan and Iain Dunning. 2011. Pulp: a linear programming toolkit for python. The University of Auckland Auckland New Zealand (2011)."},{"key":"e_1_3_3_1_46_2","unstructured":"Jayashree Mohan Amar Phanishayee Janardhan Kulkarni and Vijay Chidambaram. 2021. Synergy: Resource sensitive DNN scheduling in multi-tenant clusters. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2110.06073 (2021)."},{"key":"e_1_3_3_1_47_2","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI)","author":"Mohan Jayashree","year":"2022","unstructured":"Jayashree Mohan, Amar Phanishayee, Janardhan Kulkarni, and Vijay Chidambaram. 2022. Looking beyond GPUs for DNN scheduling on Multi-Tenant clusters. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI)."},{"key":"e_1_3_3_1_48_2","doi-asserted-by":"publisher","DOI":"10.1145\/3394885.3431535"},{"key":"e_1_3_3_1_49_2","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation (OSDI)","author":"Narayanan Deepak","year":"2020","unstructured":"Deepak Narayanan, Keshav Santhanam, Fiodar Kazhamiaka, Amar Phanishayee, and Matei Zaharia. 2020. Heterogeneity-Aware cluster scheduling policies for deep learning workloads. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI)."},{"key":"e_1_3_3_1_50_2","unstructured":"Nvidia. [n. d.]. NVIDIA H100 Tensor Core GPU. https:\/\/www.nvidia.com\/en-us\/data-center\/h100\/."},{"key":"e_1_3_3_1_51_2","volume-title":"Joint European Conference on Machine Learning and Knowledge Discovery in Databases","author":"Osterwind Adrian","year":"2022","unstructured":"Adrian Osterwind, Julian Droste-Rehling, Manoj-Rohit Vemparala, and Domenik Helms. 2022. Hardware execution time prediction for neural network layers. In Joint European Conference on Machine Learning and Knowledge Discovery in Databases. Springer."},{"key":"e_1_3_3_1_52_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651329"},{"key":"e_1_3_3_1_53_2","doi-asserted-by":"crossref","unstructured":"Pratyush Patel Zibo Gong Syeda Rizvi Esha Choukse Pulkit Misra Thomas Anderson and Akshitha Sriraman. 2023. Towards improved power management in cloud gpus. IEEE Computer Architecture Letters (2023).","DOI":"10.1109\/LCA.2023.3278652"},{"key":"e_1_3_3_1_54_2","volume-title":"USENIX Annual Technical Conference (USENIX ATC)","author":"Qiu Haoran","year":"2024","unstructured":"Haoran Qiu, Weichao Mao, Archit Patke, Shengkun Cui, Saurabh Jha, Chen Wang, Hubertus Franke, Zbigniew Kalbarczyk, Tamer Ba\u015far, and Ravishankar\u00a0K Iyer. 2024. Power-aware deep learning model serving with {\u03bc -Serve}. In USENIX Annual Technical Conference (USENIX ATC)."},{"key":"e_1_3_3_1_55_2","volume-title":"21st USENIX Symposium on Networked Systems Design and Implementation (NSDI)","author":"Rajasekaran Sudarsanan","year":"2024","unstructured":"Sudarsanan Rajasekaran, Manya Ghobadi, and Aditya Akella. 2024. CASSINI:Network-Aware job scheduling in machine learning clusters. In 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI)."},{"key":"e_1_3_3_1_56_2","doi-asserted-by":"publisher","DOI":"10.1145\/3642970.3655827"},{"key":"e_1_3_3_1_57_2","unstructured":"Anton Shilov. 2023. Nvidia\u2019s H100 GPUs will consume more power than some countries \u2014 each GPU consumes 700W of power 3.5 million are expected to be sold in the coming year. https:\/\/www.tomshardware.com\/."},{"key":"e_1_3_3_1_58_2","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3629578"},{"key":"e_1_3_3_1_59_2","doi-asserted-by":"publisher","DOI":"10.1145\/3769102.3772715"},{"key":"e_1_3_3_1_60_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPCCC66453.2025.11304653"},{"key":"e_1_3_3_1_61_2","doi-asserted-by":"crossref","unstructured":"Srinivasan Subramaniyan and Xiaorui Wang. 2025. FC-GPU: Feedback Control GPU Scheduling for Real-time Embedded Systems. ACM Transactions on Embedded Computing Systems 24 5s (2025).","DOI":"10.1145\/3761812"},{"key":"e_1_3_3_1_62_2","doi-asserted-by":"publisher","DOI":"10.1145\/3307772.3328315"},{"key":"e_1_3_3_1_63_2","doi-asserted-by":"crossref","unstructured":"Balavignesh Vemparala Ming Yang and Soheil Soghrati. 2024. Deep learning-driven domain decomposition (DLD3): A generalizable AI-driven framework for structural analysis. Computer Methods in Applied Mechanics and Engineering 432 (2024).","DOI":"10.1016\/j.cma.2024.117446"},{"key":"e_1_3_3_1_64_2","unstructured":"Balavignesh Vemparala Narayana\u00a0Murthy. 2024. Advanced Computational and Deep Learning Techniques for Modeling Materials with Complex Microstructures. Ph.\u00a0D. Dissertation. The Ohio State University."},{"key":"e_1_3_3_1_65_2","volume-title":"Proceedings of the conference on USENIX Annual technical conference (USENIX ATC)","author":"Verma Akshat","year":"2009","unstructured":"Akshat Verma, Gargi Dasgupta, Tapan\u00a0Kumar Nayak, Pradipta De, and Ravi Kothari. 2009. Server workload analysis for power minimization using consolidation. In Proceedings of the conference on USENIX Annual technical conference (USENIX ATC)."},{"key":"e_1_3_3_1_66_2","unstructured":"Farui Wang Weizhe Zhang Shichao Lai Meng Hao and Zheng Wang. 2021. Dynamic GPU energy optimization for machine learning training workloads. IEEE Transactions on Parallel and Distributed Systems 33 11 (2021)."},{"key":"e_1_3_3_1_67_2","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC47752.2019.9042047"},{"key":"e_1_3_3_1_68_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2008.4658631"},{"key":"e_1_3_3_1_69_2","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2009.34"},{"key":"e_1_3_3_1_70_2","doi-asserted-by":"publisher","DOI":"10.1109\/INFCOM.2012.6195471"},{"key":"e_1_3_3_1_71_2","volume-title":"USENIX Annual Technical Conference (USENIX ATC)","author":"Weng Qizhen","year":"2023","unstructured":"Qizhen Weng, Lingyun Yang, Yinghao Yu, Wei Wang, Xiaochuan Tang, Guodong Yang, and Liping Zhang. 2023. Beware of Fragmentation: Scheduling GPU-Sharing Workloads with Fragmentation Gradient Descent. In USENIX Annual Technical Conference (USENIX ATC)."},{"key":"e_1_3_3_1_72_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01548"},{"key":"e_1_3_3_1_73_2","doi-asserted-by":"crossref","unstructured":"Zhisheng Ye Wei Gao Qinghao Hu Peng Sun Xiaolin Wang Yingwei Luo Tianwei Zhang and Yonggang Wen. 2024. Deep learning workload scheduling in gpu datacenters: A survey. Comput. Surveys 56 6 (2024).","DOI":"10.1145\/3638757"},{"key":"e_1_3_3_1_74_2","doi-asserted-by":"publisher","DOI":"10.5555\/3485849.3485855"},{"key":"e_1_3_3_1_75_2","doi-asserted-by":"publisher","DOI":"10.1145\/3605573.3605609"},{"key":"e_1_3_3_1_76_2","doi-asserted-by":"crossref","unstructured":"Sheng Zhang Zhuzhong Qian Zhaoyi Luo Jie Wu and Sanglu Lu. 2015. Burstiness-aware resource reservation for server consolidation in computing clouds. IEEE Transactions on Parallel and Distributed Systems 27 4 (2015).","DOI":"10.1109\/TPDS.2015.2425403"}],"event":{"name":"ICS '26: 2026 International Conference on Supercomputing","location":"Belfast United Kingdom","acronym":"ICS '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 40th ACM International Conference on Supercomputing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/abs\/10.1145\/3797905.3800546","content-type":"text\/html","content-version":"vor","intended-application":"syndication"}],"deposited":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T12:57:21Z","timestamp":1782997041000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3797905.3800546"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":75,"alternative-id":["10.1145\/3797905.3800546","10.1145\/3797905"],"URL":"https:\/\/doi.org\/10.1145\/3797905.3800546","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}