{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,15]],"date-time":"2026-08-15T17:43:15Z","timestamp":1786815795110,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":74,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,4,27]],"date-time":"2024-04-27T00:00:00Z","timestamp":1714176000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CNS-2104548"],"award-info":[{"award-number":["CNS-2104548"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100016682","name":"VMware","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100016682","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,4,27]]},"DOI":"10.1145\/3620666.3651329","type":"proceedings-article","created":{"date-parts":[[2024,4,24]],"date-time":"2024-04-24T12:08:21Z","timestamp":1713960501000},"page":"207-222","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":82,"title":["Characterizing Power Management Opportunities for LLMs in the Cloud"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3611-5160","authenticated-orcid":false,"given":"Pratyush","family":"Patel","sequence":"first","affiliation":[{"name":"Microsoft Azure, Redmond, USA"},{"name":"University of Washington, Seattle, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0371-5522","authenticated-orcid":false,"given":"Esha","family":"Choukse","sequence":"additional","affiliation":[{"name":"Microsoft Azure, Redmond, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-8334-1291","authenticated-orcid":false,"given":"Chaojie","family":"Zhang","sequence":"additional","affiliation":[{"name":"Microsoft Azure, Redmond, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2591-4012","authenticated-orcid":false,"given":"\u00cd\u00f1igo","family":"Goiri","sequence":"additional","affiliation":[{"name":"Microsoft Azure, Redmond, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-4665-2319","authenticated-orcid":false,"given":"Brijesh","family":"Warrier","sequence":"additional","affiliation":[{"name":"Microsoft Azure, Redmond, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-1079-2646","authenticated-orcid":false,"given":"Nithish","family":"Mahalingam","sequence":"additional","affiliation":[{"name":"Microsoft Azure, Redmond, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5971-5084","authenticated-orcid":false,"given":"Ricardo","family":"Bianchini","sequence":"additional","affiliation":[{"name":"Microsoft Azure, Redmond, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,4,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"https:\/\/aws.amazon.com\/sagemaker","author":"SageMaker Amazon","year":"2023","unstructured":"Amazon SageMaker. https:\/\/aws.amazon.com\/sagemaker, 2023."},{"key":"e_1_3_2_1_2_1","volume-title":"https:\/\/azure.microsoft.com\/en-us\/products\/machine-learning","author":"Service Azure Machine","year":"2023","unstructured":"Azure Machine Learning - ML as a Service. https:\/\/azure.microsoft.com\/en-us\/products\/machine-learning, 2023."},{"key":"e_1_3_2_1_3_1","unstructured":"AMD. ROCm Open Software Platform for GPU Compute. https:\/\/www.amd.com\/en\/graphics\/servers-solutions-rocm."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Reza Yazdani Aminabadi Samyam Rajbhandari Ammar Ahmad Awan Cheng Li Du Li Elton Zheng Olatunji Ruwase Shaden Smith Minjia Zhang Jeff Rasley and Yuxiong He. DeepSpeed-Inference: Enabling Efficient Inference of Transformer Models at Unprecedented Scale. In SC 2022.","DOI":"10.1109\/SC41404.2022.00051"},{"key":"e_1_3_2_1_5_1","volume-title":"GPT-NeoX: Large Scale Autoregressive Language Modeling in PyTorch. https:\/\/www.github.com\/eleutherai\/gpt-neox","author":"Andonian Alex","year":"2021","unstructured":"Alex Andonian, Quentin Anthony, Stella Biderman, Sid Black, Preetham Gali, Leo Gao, Eric Hallahan, Josh Levy-Kramer, Connor Leahy, Lucas Nestler, Kip Parker, Michael Pieler, Shivanshu Purohit, Tri Songz, Wang Phil, and Samuel Weinbach. GPT-NeoX: Large Scale Autoregressive Language Modeling in PyTorch. https:\/\/www.github.com\/eleutherai\/gpt-neox, 2021."},{"key":"e_1_3_2_1_6_1","volume-title":"Azure OpenAI Service. https:\/\/azure.microsoft.com\/en-us\/products\/ai-services\/openai-service","author":"Azure Microsoft","year":"2022","unstructured":"Microsoft Azure. Azure OpenAI Service. https:\/\/azure.microsoft.com\/en-us\/products\/ai-services\/openai-service, 2022."},{"key":"e_1_3_2_1_7_1","volume-title":"Synthesis Lectures on Computer Architecture","author":"Barroso Luiz Andr\u00e9","year":"2018","unstructured":"Luiz Andr\u00e9 Barroso, Urs H\u00f6lzle, and Parthasarathy Ranganathan. The Datacenter as a Computer: Designing Warehouse-Scale Machines. Synthesis Lectures on Computer Architecture, 2018."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442188.3445922"},{"key":"e_1_3_2_1_9_1","volume-title":"ASPLOS","author":"Bharadwaj Srikant","year":"2023","unstructured":"Srikant Bharadwaj, Shomit Das, Kaushik Mazumdar, Bradford Beckmann, and Stephen Kosonocky. Predict; Do Not React for Enabling Efficient Fine Grain DVFS in GPUs. In ASPLOS, 2023."},{"key":"e_1_3_2_1_10_1","volume-title":"NeurIPS","author":"Brown Tom B.","year":"2020","unstructured":"Tom B. Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, Sandhini Agarwal, Ariel Herbert-Voss, Gretchen Krueger, Tom Henighan, Rewon Child, Aditya Ramesh, Daniel M. Ziegler, Jeffrey Wu, Clemens Winter, Christopher Hesse, Mark Chen, Eric Sigler, Mateusz Litwin, Scott Gray, Benjamin Chess, Jack Clark, Christopher Berner, Sam McCandlish, Alec Radford, Ilya Sutskever, and Dario Amodei. Language Models Are Few-Shot Learners. In NeurIPS, 2020."},{"key":"e_1_3_2_1_11_1","volume-title":"USENIX ATC","author":"Choi Sangjin","year":"2023","unstructured":"Sangjin Choi, Inhoe Koo, Jeongseob Ahn, Myeongjae Jeon, and Youngjin Kwon. EnvPipe: Performance-preserving DNN Training Framework for Saving Energy. In USENIX ATC, 2023."},{"key":"e_1_3_2_1_12_1","volume-title":"Scaling Instruction-Finetuned Language Models. arXiv preprint arXiv:2210.11416","author":"Chung Hyung Won","year":"2022","unstructured":"Hyung Won Chung, Le Hou, Shayne Longpre, Barret Zoph, Yi Tay, William Fedus, Eric Li, Xuezhi Wang, Mostafa Dehghani, Siddhartha Brahma, et al. Scaling Instruction-Finetuned Language Models. arXiv preprint arXiv:2210.11416, 2022."},{"key":"e_1_3_2_1_13_1","volume-title":"NeurIPS","author":"Dettmers Tim","year":"2022","unstructured":"Tim Dettmers, Mike Lewis, Younes Belkada, and Luke Zettlemoyer. LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale. In NeurIPS, 2022."},{"key":"e_1_3_2_1_14_1","volume-title":"NAACL","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In NAACL, 2019."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/1250662.1250665"},{"key":"e_1_3_2_1_16_1","volume-title":"Strategic Comments","author":"International Institute for Strategic Studies.","year":"2023","unstructured":"International Institute for Strategic Studies. Large Language Models: Fast Proliferation and Budding International Competition. Strategic Comments, 2023."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/1998582.1998589"},{"key":"e_1_3_2_1_18_1","volume-title":"bitsandbytes: Memory decreases but latency increases. https:\/\/github.com\/TimDettmers\/bitsandbytes\/issues\/6","year":"2022","unstructured":"GitHub. bitsandbytes: Memory decreases but latency increases. https:\/\/github.com\/TimDettmers\/bitsandbytes\/issues\/6, 2022."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/1519065.1519099"},{"key":"e_1_3_2_1_20_1","volume-title":"Accelerate: Training and Inference at Scale Made Simple, Efficient and Adaptable. https:\/\/github.com\/huggingface\/accelerate","author":"Gugger Sylvain","year":"2022","unstructured":"Sylvain Gugger, Lysandre Debut, Thomas Wolf, Philipp Schmid, Zachary Mueller, and Sourab Mangrulkar. Accelerate: Training and Inference at Scale Made Simple, Efficient and Adaptable. https:\/\/github.com\/huggingface\/accelerate, 2022."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2018.00059"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/BigData47090.2019.9005632"},{"key":"e_1_3_2_1_23_1","volume-title":"ASPLOS","author":"Hsu Chang-Hong","year":"2018","unstructured":"Chang-Hong Hsu, Qingyuan Deng, Jason Mars, and Lingjia Tang. SmoothOperator: Reducing Power Fragmentation and Improving Power Utilization in Large-Scale Datacenters. In ASPLOS, 2018."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476223"},{"key":"e_1_3_2_1_25_1","volume-title":"NeurIPS","author":"Huang Yanping","year":"2019","unstructured":"Yanping Huang, Youlong Cheng, Ankur Bapna, Orhan Firat, Dehao Chen, Mia Chen, HyoukJoong Lee, Jiquan Ngiam, Quoc V Le, Yonghui Wu, et al. Gpipe: Efficient Training of Giant Neural Networks Using Pipeline Parallelism. In NeurIPS, 2019."},{"key":"e_1_3_2_1_26_1","volume-title":"Intelligent Platform Management Interface Specification (IPMI). https:\/\/www.intel.in\/content\/www\/in\/en\/products\/docs\/servers\/ipmi\/ipmi-second-gen-interface-spec-v2-rev1-1.html","author":"Packard Hewlett","year":"2013","unstructured":"Intel, Hewlett Packard, NEC, and Dell. Intelligent Platform Management Interface Specification (IPMI). https:\/\/www.intel.in\/content\/www\/in\/en\/products\/docs\/servers\/ipmi\/ipmi-second-gen-interface-spec-v2-rev1-1.html, 2013."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2020.3023723"},{"key":"e_1_3_2_1_28_1","volume-title":"ISCA","author":"Jalili Majid","year":"2021","unstructured":"Majid Jalili, Ioannis Manousakis, \u00cd\u00f1igo Goiri, Pulkit A. Misra, Ashish Raniwala, Husam Alissa, Bharath Ramakrishnan, Phillip Tuma, Christian Belady, Marcus Fontoura, and Ricardo Bianchini. Cost-Efficient Overclocking in Immersion-Cooled Datacenters. In ISCA, 2021."},{"key":"e_1_3_2_1_29_1","volume-title":"USENIX ATC","author":"Jeon Myeongjae","year":"2019","unstructured":"Myeongjae Jeon, Shivaram Venkataraman, Amar Phanishayee, Junjie Qian, Wencong Xiao, and Fan Yang. Analysis of Large-Scale Multi-Tenant GPU clusters for DNN Training Workloads. In USENIX ATC, 2019."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3177754"},{"key":"e_1_3_2_1_31_1","volume-title":"USENIX ATC","author":"Kumbhare Alok Gautam","year":"2021","unstructured":"Alok Gautam Kumbhare, Reza Azimi, Ioannis Manousakis, Anand Bonde, Felipe Frujeri, Nithish Mahalingam, Pulkit A. Misra, Seyyed Ahmad Javadi, Bianca Schroeder, Marcus Fontoura, and Ricardo Bianchini. Prediction-Based Power Oversubscription in Cloud Platforms. In USENIX ATC, 2021."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00093"},{"key":"e_1_3_2_1_34_1","volume-title":"OSDI","author":"Li Shaohong","year":"2020","unstructured":"Shaohong Li, Xi Wang, Xiao Zhang, Vasileios Kontorinis, Sreekumar Kodakara, David Lo, and Parthasarathy Ranganathan. Thunderbolt: Throughput-Optimized, Quality-of-Service-Aware Power Capping at Scale. In OSDI, 2020."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.14778\/3415478.3415530"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2019.00067"},{"key":"e_1_3_2_1_37_1","volume-title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach. arXiv preprint arXiv:1907.11692","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov. RoBERTa: A Robustly Optimized BERT Pretraining Approach. arXiv preprint arXiv:1907.11692, 2019."},{"key":"e_1_3_2_1_38_1","unstructured":"Meta. Introducing the AI Research SuperCluster --- Meta's Cutting-edge AI Supercomputer for AI Research. https:\/\/ai.facebook.com\/blog\/ai-rsc\/."},{"key":"e_1_3_2_1_39_1","unstructured":"Microsoft. DeepSpeed: Model Implementations for Inference (MII). https:\/\/github.com\/microsoft\/DeepSpeed-MII."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2022.3144614"},{"key":"e_1_3_2_1_41_1","volume-title":"MASCOTS","author":"Nie Bin","year":"2017","unstructured":"Bin Nie, Ji Xue, Saurabh Gupta, Christian Engelmann, Evgenia Smirni, and Devesh Tiwari. Characterizing Temperature, Power, and Soft-error Behaviors in Data Center Systems: Insights, Challenges, and Opportunities. In MASCOTS, 2017."},{"key":"e_1_3_2_1_42_1","unstructured":"NVIDIA. Data Center GPU Driver. https:\/\/docs.nvidia.com\/datacenter\/tesla\/pdf\/NVIDIA_Data_Center_GPU_Driver_Release_Notes_450_v1.pdf."},{"key":"e_1_3_2_1_43_1","unstructured":"NVIDIA. Data Center GPU Manager (DCGM). https:\/\/developer.nvidia.com\/dcgm."},{"key":"e_1_3_2_1_44_1","unstructured":"NVIDIA. DGX A100: The Universal System for AI Infrastructure. https:\/\/resources.nvidia.com\/en-us-dgx-systems\/dgx-ai."},{"key":"e_1_3_2_1_45_1","unstructured":"NVIDIA. DGX H100. https:\/\/www.nvidia.com\/en-us\/data-center\/dgx-h100\/."},{"key":"e_1_3_2_1_46_1","unstructured":"NVIDIA. NVIDIA A100 80GB PCIe GPU Product Brief. https:\/\/www.nvidia.com\/content\/dam\/en-zz\/Solutions\/Data-Center\/a100\/pdf\/PB-10577-001_v02.pdf."},{"key":"e_1_3_2_1_47_1","unstructured":"NVIDIA. System Management Interface (nvidia-smi). https:\/\/developer.nvidia.com\/nvidia-system-management-interface."},{"key":"e_1_3_2_1_48_1","unstructured":"OpenAI. Scaling Kubernetes to 7 500 Nodes. https:\/\/openai.com\/research\/scaling-kubernetes-to-7500-nodes."},{"key":"e_1_3_2_1_49_1","volume-title":"Splitwise: Efficient Generative LLM Inference Using Phase Splitting. arXiv preprint arXiv:2311.18677","author":"Patel Pratyush","year":"2023","unstructured":"Pratyush Patel, Esha Choukse, Chaojie Zhang, \u00cd\u00f1igo Goiri, Aashaka Shah, Saeed Maleki, and Ricardo Bianchini. Splitwise: Efficient Generative LLM Inference Using Phase Splitting. arXiv preprint arXiv:2311.18677, 2023."},{"key":"e_1_3_2_1_50_1","volume-title":"POLCA: Power Oversubscription in LLM Cloud Providers. arXiv preprint arXiv:2308.12908","author":"Patel Pratyush","year":"2023","unstructured":"Pratyush Patel, Esha Choukse, Chaojie Zhang, \u00cd\u00f1igo Goiri, Brijesh Warrier, Nithish Mahalingam, and Ricardo Bianchini. POLCA: Power Oversubscription in LLM Cloud Providers. arXiv preprint arXiv:2308.12908, 2023."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2023.3278652"},{"key":"e_1_3_2_1_52_1","volume-title":"WORK","author":"Patki Tapasya","year":"2019","unstructured":"Tapasya Patki, Zachary Frye, Harsh Bhatia, Francesco Di Natale, James Glosli, Helgi Ingolfsson, and Barry Rountree. Comparing GPU Power and Frequency Capping: A Case Study with the MuMMI Workflow. In WORK, 2019."},{"key":"e_1_3_2_1_53_1","volume-title":"Then Shrink. Computer","author":"Patterson David","year":"2022","unstructured":"David Patterson, Joseph Gonzalez, Urs H\u00f6lzle, Quoc Le, Chen Liang, Lluis-Miquel Munguia, Daniel Rothchild, David R So, Maud Texier, and Jeff Dean. The Carbon Footprint of Machine Learning Training Will Plateau, Then Shrink. Computer, 2022."},{"key":"e_1_3_2_1_54_1","volume-title":"XDC","author":"Peres Martin","year":"2013","unstructured":"Martin Peres. Reverse Engineering Power Management on NVIDIA GPUs - A Detailed Overview. In XDC, 2013."},{"key":"e_1_3_2_1_55_1","volume-title":"What Does Not. In ICPADS","author":"Petoumenos Pavlos","year":"2015","unstructured":"Pavlos Petoumenos, Lev Mukhanov, Zheng Wang, Hugh Leather, and Dimitrios S. Nikolopoulos. Power Capping: What Works, What Does Not. In ICPADS, 2015."},{"key":"e_1_3_2_1_56_1","volume-title":"https:\/\/cloud.google.com\/vertex-ai","author":"Platform Google Cloud","year":"2023","unstructured":"Google Cloud Platform. Vertex AI. https:\/\/cloud.google.com\/vertex-ai, 2023."},{"key":"e_1_3_2_1_57_1","volume-title":"Language Models are Unsupervised Multitask Learners","author":"Radford Alec","year":"2019","unstructured":"Alec Radford, Jeffrey Wu, Rewon Child, David Luan, Dario Amodei, and Ilya Sutskever. Language Models are Unsupervised Multitask Learners, 2019."},{"key":"e_1_3_2_1_58_1","volume-title":"ISCA","author":"Ranganathan Parthasarathy","year":"2006","unstructured":"Parthasarathy Ranganathan, Phil Leech, David Irwin, and Jeffrey Chase. Ensemble-level Power Management for Dense Blade Servers. In ISCA, 2006."},{"key":"e_1_3_2_1_59_1","volume-title":"Why Your AI infrastructure Needs Both Training and Inference","author":"Research Tirias","year":"2019","unstructured":"Tirias Research. Why Your AI infrastructure Needs Both Training and Inference. 2019."},{"key":"e_1_3_2_1_60_1","unstructured":"Philipp Schmid. Fine-tune FLAN-T5 XL\/XXL using DeepSpeed & Hugging Face Transformers. https:\/\/www.philschmid.de\/fine-tune-flan-t5-deepspeed."},{"key":"e_1_3_2_1_61_1","unstructured":"Amazon Web Services. Amazon EC2 Update - Inf1 Instances with AWS Inferentia Chips for High Performance Cost-Effective Inferencing. https:\/\/aws.amazon.com\/blogs\/aws\/amazon-ec2-update-inf1-instances-with-aws-inferentia-chips-for-high-performance-cost-effective-inferencing\/."},{"key":"e_1_3_2_1_62_1","unstructured":"Amazon Web Services. AWS Trainium: High-performance Machine Learning Training Accelerator Purpose Built by AWS. https:\/\/aws.amazon.com\/machine-learning\/trainium\/."},{"key":"e_1_3_2_1_63_1","volume-title":"Megatron-LM: Training Multi-billion Parameter Language Models Using Model Parallelism. arXiv preprint arXiv:1909.08053","author":"Shoeybi Mohammad","year":"2019","unstructured":"Mohammad Shoeybi, Mostofa Patwary, Raul Puri, Patrick LeGresley, Jared Casper, and Bryan Catanzaro. Megatron-LM: Training Multi-billion Parameter Language Models Using Model Parallelism. arXiv preprint arXiv:1909.08053, 2019."},{"key":"e_1_3_2_1_64_1","volume-title":"Planet-scale, Preemptive and Elastic Scheduling of AI Workloads. arXiv preprint arXiv:2202.07848","author":"Shukla Dharma","year":"2022","unstructured":"Dharma Shukla, Muthian Sivathanu, Srinidhi Viswanatha, Bhargav Gulavani, Rimma Nehme, Amey Agrawal, Chen Chen, Nipun Kwatra, Ramachandran Ramjee, Pankaj Sharma, Atul Katiyar, Vipul Modi, Vaibhav Sharma, Abhishek Singh, Shreshth Singhal, Kaustubh Welankar, Lu Xun, Ravi Anupindi, Karthik Elangovan, and Mark Russinovich. Singularity: Planet-scale, Preemptive and Elastic Scheduling of AI Workloads. arXiv preprint arXiv:2202.07848, 2022."},{"key":"e_1_3_2_1_65_1","volume-title":"Accelerator-rich Systems. In SC","author":"Sinha Prasoon","year":"2022","unstructured":"Prasoon Sinha, Akhil Guliani, Rutwik Jain, Brandon Tran, Matthew D Sinclair, and Shivaram Venkataraman. Not All GPUs Are Created Equal: Characterizing Variability in Large-scale, Accelerator-rich Systems. In SC, 2022."},{"key":"e_1_3_2_1_66_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale Dan Bikel Lukas Blecher Cristian Canton Ferrer Moya Chen Guillem Cucurull David Esiobu Jude Fernandes Jeremy Fu Wenyin Fu Brian Fuller Cynthia Gao Vedanuj Goswami Naman Goyal Anthony Hartshorn Saghar Hosseini Rui Hou Hakan Inan Marcin Kardas Viktor Kerkez Madian Khabsa Isabel Kloumann Artem Korenev Punit Singh Koura Marie-Anne Lachaux Thibaut Lavril Jenya Lee Diana Liskovich Yinghai Lu Yuning Mao Xavier Martinet Todor Mihaylov Pushkar Mishra Igor Molybog Yixin Nie Andrew Poulton Jeremy Reizenstein Rashi Rungta Kalyan Saladi Alan Schelten Ruan Silva Eric Michael Smith Ranjan Subramanian Xiaoqing Ellen Tan Binh Tang Ross Taylor Adina Williams Jian Xiang Kuan Puxin Xu Zheng Yan Iliyan Zarov Yuchen Zhang Angela Fan Melanie Kambadur Sharan Narang Aurelien Rodriguez Robert Stojnic Sergey Edunov and Thomas Scialom. Llama 2: Open Foundation and Fine-tuned Chat Models. arXiv preprint arXiv:2307.09288 2023."},{"key":"e_1_3_2_1_67_1","volume-title":"NeurIPS","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. Attention is All You Need. In NeurIPS, 2017."},{"key":"e_1_3_2_1_68_1","volume-title":"HPC","author":"Vu Lan","year":"2014","unstructured":"Lan Vu, Hari Sivaraman, and Rishi Bidarkar. GPU Virtualization for High Performance General Purpose Computing on the ESX Hypervisor. In HPC, 2014."},{"key":"e_1_3_2_1_69_1","volume-title":"Transformers: State-of-the-art Natural Language Processing. In EMNLP","author":"Wolf Thomas","year":"2020","unstructured":"Thomas Wolf, Lysandre Debut, Victor Sanh, Julien Chaumond, Clement Delangue, Anthony Moi, Pierric Cistac, Tim Rault, R\u00e9mi Louf, Morgan Funtowicz, Joe Davison, Sam Shleifer, Patrick von Platen, Clara Ma, Yacine Jernite, Julien Plu, Canwen Xu, Teven Le Scao, Sylvain Gugger, Mariama Drame, Quentin Lhoest, and Alexander M. Rush. Transformers: State-of-the-art Natural Language Processing. In EMNLP, 2020."},{"key":"e_1_3_2_1_70_1","unstructured":"BigScience Workshop Teven Le Scao Angela Fan Christopher Akiki Ellie Pavlick Suzana Ili\u0107 Daniel Hesslow Roman Castagn\u00e9 Alexandra Sasha Luccioni Fran\u00e7ois Yvon Matthias Gall\u00e9 Jonathan Tow Alexander M. Rush Stella Biderman Albert Webson Pawan Sasanka Ammanamanchi Thomas Wang Beno\u00eet Sagot Niklas Muennighoff Albert Villanova del Moral Olatunji Ruwase Rachel Bawden Stas Bekman Angelina McMillan-Major Iz Beltagy Huu Nguyen Lucile Saulnier Samson Tan Pedro Ortiz Suarez Victor Sanh Hugo Lauren\u00e7on Yacine Jernite Julien Launay Margaret Mitchell and Colin Raffel. BLOOM: A 176B-Parameter Open-access Multilingual Language Model. arXiv preprint arXiv:2211.05100 2022."},{"key":"e_1_3_2_1_71_1","volume-title":"NSDI","author":"You Jie","year":"2023","unstructured":"Jie You, Jae-Won Chung, and Mosharaf Chowdhury. Zeus: Understanding and Optimizing GPU Energy Consumption of DNN Training. In NSDI, 2023."},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10070943"},{"key":"e_1_3_2_1_73_1","volume-title":"ISCA","author":"Zhang Chaojie","year":"2021","unstructured":"Chaojie Zhang, Alok Gautam Kumbhare, Ioannis Manousakis, Deli Zhang, Pulkit A Misra, Rod Assis, Kyle Woolcock, Nithish Mahalingam, Brijesh Warrier, David Gauthier, Lalu Kunnath, Steve Solomon, Osvaldo Morales, Marcus Fontoura, and Ricardo Bianchini. Flex: High-Availability Datacenters With Zero Reserved Power. In ISCA, 2021."},{"key":"e_1_3_2_1_74_1","unstructured":"Susan Zhang Stephen Roller Naman Goyal Mikel Artetxe Moya Chen Shuohui Chen Christopher Dewan Mona Diab Xian Li Xi Victoria Lin Todor Mihaylov Myle Ott Sam Shleifer Kurt Shuster Daniel Simig Punit Singh Koura Anjali Sridhar Tianlu Wang and Luke Zettlemoyer. OPT: Open Pre-trained Transformer Language Models. arXiv preprint arXiv:2205.01068 2022."}],"event":{"name":"ASPLOS '24: 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 3","location":"La Jolla CA USA","acronym":"ASPLOS '24","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture","SIGOPS ACM Special Interest Group on Operating Systems","SIGPLAN ACM Special Interest Group on Programming Languages","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 3"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3620666.3651329","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/abs\/10.1145\/3620666.3651329","content-type":"text\/html","content-version":"vor","intended-application":"syndication"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:03:42Z","timestamp":1750291422000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3620666.3651329"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,27]]},"references-count":74,"alternative-id":["10.1145\/3620666.3651329","10.1145\/3620666"],"URL":"https:\/\/doi.org\/10.1145\/3620666.3651329","relation":{},"subject":[],"published":{"date-parts":[[2024,4,27]]},"assertion":[{"value":"2024-04-27","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}