{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T13:43:19Z","timestamp":1782999799283,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,6]]},"DOI":"10.1145\/3797905.3815788","type":"proceedings-article","created":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T11:50:37Z","timestamp":1782993037000},"page":"1128-1140","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Phase-aware Peak Power Reduction for Minimizing the Capital Expense of LLM Inference"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0805-1916","authenticated-orcid":false,"given":"Simon","family":"Wu","sequence":"first","affiliation":[{"name":"Electrical and Computer Engineering, The Ohio State University, Columbus, Ohio, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-8028-7113","authenticated-orcid":false,"given":"Yuan","family":"Ma","sequence":"additional","affiliation":[{"name":"Electrical and Computer Engineering, The Ohio State University, Columbus, Ohio, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9633-1418","authenticated-orcid":false,"given":"Xiaorui","family":"Wang","sequence":"additional","affiliation":[{"name":"Electrical and Computer Engineering, The Ohio State University, Colunbus, Ohio, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,5]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"2021. Scaling Kubernetes to 7 500 nodes. https:\/\/openai.com\/index\/scaling-kubernetes-to-7500-nodes\/"},{"key":"e_1_3_3_1_3_2","unstructured":"2022. Introducing the AI Research SuperCluster \u2014 Meta\u2019s cutting-edge AI supercomputer for AI research. https:\/\/ai.meta.com\/blog\/ai-rsc\/"},{"key":"e_1_3_3_1_4_2","unstructured":"2026. Extended Results for Phase-aware Peak Power Reduction for Minimizing the Capital Expense of LLM Inference. https:\/\/www.dropbox.com\/scl\/fi\/d63rydn9m0s8c4j7tb2sk\/ICS2026_PPPR_Shepherd_extend.pdf?rlkey=idhf2k80rqvow2vs3u3dc3szs&st=cfn4pwt9&dl=0 Extended version."},{"key":"e_1_3_3_1_5_2","first-page":"117","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Agrawal Amey","year":"2024","unstructured":"Amey Agrawal, Nitin Kedia, Ashish Panwar, Jayashree Mohan, Nipun Kwatra, Bhargav Gulavani, Alexey Tumanov, and Ramachandran Ramjee. 2024. Taming Throughput-Latency tradeoff in LLM inference with Sarathi-Serve. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24). 117\u2013134."},{"key":"e_1_3_3_1_6_2","unstructured":"anon8231489123. 2023. ShareGPT Vicuna unfiltered: ShareGPT_V3_unfiltered_cleaned_split.json. https:\/\/huggingface.co\/datasets\/anon8231489123\/ShareGPT_Vicuna_unfiltered. Accessed 2025-11-09."},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-01761-2"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS60910.2024.00051"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCEP.2009.5212025"},{"key":"e_1_3_3_1_10_2","unstructured":"An\u00a0Yang et al.2024. Qwen2 Technical Report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.10671 (2024)."},{"key":"e_1_3_3_1_11_2","unstructured":"Hugo\u00a0Touvron et al.2023. Llama 2: Open Foundation and Fine-Tuned Chat Models. arxiv:https:\/\/arXiv.org\/abs\/2307.09288\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2307.09288"},{"key":"e_1_3_3_1_12_2","unstructured":"Ivan Goldwasser Harry Petty Pradyumna Desale and Kirthi Devleker. 2024. NVIDIA GB200 NVL72 Delivers Trillion-Parameter LLM Training and Real-Time Inference. https:\/\/developer.nvidia.com\/blog\/nvidia-gb200-nvl72-delivers-trillion-parameter-llm-training-and-real-time-inference\/. Accessed: 2026-01-31."},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/1519065.1519099"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"publisher","DOI":"10.1145\/2000064.2000105"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/2150976.2150985"},{"key":"e_1_3_3_1_16_2","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et\u00a0al. 2024. The llama 3 herd of models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.21783 (2024). https:\/\/arxiv.org\/pdf\/2407.21783"},{"key":"e_1_3_3_1_17_2","volume-title":"Text Generation Inference (TGI)","author":"Face Hugging","year":"2025","unstructured":"Hugging Face. 2025. Text Generation Inference (TGI). https:\/\/github.com\/huggingface\/text-generation-inference commit efb94e0; accessed 2025-11-07."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICAC.2019.00015"},{"key":"e_1_3_3_1_19_2","unstructured":"Albert\u00a0Q. Jiang Alexandre Sablayrolles Arthur Mensch Chris Bamford Devendra\u00a0Singh Chaplot Diego de\u00a0las Casas Florian Bressand Gianna Lengyel Guillaume Lample Lucile Saulnier L\u00e9lio\u00a0Renard Lavaud Marie-Anne Lachaux Pierre Stock Teven\u00a0Le Scao Thibaut Lavril Thomas Wang Timoth\u00e9e Lacroix and William\u00a0El Sayed. 2023. Mistral 7B. arxiv:https:\/\/arXiv.org\/abs\/2310.06825\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2310.06825"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/3676641.3715996"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Vasileios Kontorinis Liuyi\u00a0Eric Zhang Baris Aksanli Jack Sampson Houman Homayoun Eddie Pettis Dean\u00a0M Tullsen and Tajana\u00a0Simunic Rosing. 2012. Managing distributed ups energy for effective power capping in data centers. ACM SIGARCH Computer Architecture News 40 3 (2012) 488\u2013499.","DOI":"10.1145\/2366231.2337216"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750384"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/IGCC.2015.7393690"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3754598.3754670"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2017.8057205"},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/2391229.2391240"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651329"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.1145\/1736020.1736047"},{"key":"e_1_3_3_1_31_2","unstructured":"Jovan Stojkovic Esha Choukse Chaojie Zhang Inigo Goiri and Josep Torrellas. 2024. Towards greener llms: Bringing energy-efficiency to the forefront of llm inference. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.20306 (2024)."},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA61900.2025.00102"},{"key":"e_1_3_3_1_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPCCC66453.2025.11304653"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","DOI":"10.1145\/3315508.3329973"},{"key":"e_1_3_3_1_35_2","unstructured":"Tripp Lite. n. d.. SMART1500LCD SmartPro LCD 120V 1500VA 900W Line-Interactive UPS Datasheet. http:\/\/images.salsify.com\/image\/upload\/s\u2013oMTfEGGZ\u2013\/53ea2bc2a4d8e5790b069ae439a9e03cb629e205.pdf. Accessed: 2026-04-17."},{"key":"e_1_3_3_1_36_2","unstructured":"vLLM Contributors. 2026. Qwen3-VL Usage Guide. vLLM Recipes Documentation. https:\/\/docs.vllm.ai\/projects\/recipes\/en\/latest\/Qwen\/Qwen3-VL.html Accessed: 2026-01-31."},{"key":"e_1_3_3_1_37_2","doi-asserted-by":"publisher","DOI":"10.1145\/2254756.2254780"},{"key":"e_1_3_3_1_38_2","doi-asserted-by":"crossref","unstructured":"Qiang Wang Xinxin Mei Hai Liu Yiu-Wing Leung Zongpeng Li and Xiaowen Chu. 2022. Energy-aware non-preemptive task scheduling with deadline constraint in DVFS-enabled heterogeneous clusters. IEEE Transactions on Parallel and Distributed Systems 33 12 (2022) 4083\u20134099.","DOI":"10.1109\/TPDS.2022.3181096"},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"publisher","DOI":"10.1145\/1996130.1996170"},{"key":"e_1_3_3_1_40_2","volume-title":"Proceedings of the 8th International Conference on Network and Service Management (CNSM)","author":"Zhang Yanwei","year":"2012","unstructured":"Yanwei Zhang, Yefu Wang, and Xiaorui Wang. 2012. TEStore: Exploiting Thermal and Energy Storage to Cut the Electricity Bill for Datacenter Cooling. In Proceedings of the 8th International Conference on Network and Service Management (CNSM)."},{"key":"e_1_3_3_1_41_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2014.6835924"},{"key":"e_1_3_3_1_42_2","doi-asserted-by":"crossref","unstructured":"Wenli Zheng Kai Ma and Xiaorui Wang. 2015. TE-Shave: Reducing data center capital and operating expenses with thermal energy storage. IEEE Trans. Comput. 64 11 (2015) 3278\u20133292.","DOI":"10.1109\/TC.2015.2394381"},{"key":"e_1_3_3_1_43_2","doi-asserted-by":"crossref","unstructured":"Wenli Zheng Kai Ma and Xiaorui Wang. 2017. Hybrid Energy Storage with Supercapacitor for Cost-Efficient Data Center Power Shaving and Capping. IEEE Transactions on Parallel and Distributed Systems 28 4 (2017) 1105\u20131118.","DOI":"10.1109\/TPDS.2016.2607715"},{"key":"e_1_3_3_1_44_2","first-page":"193","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Zhong Yinmin","year":"2024","unstructured":"Yinmin Zhong, Shengyu Liu, Junda Chen, Jianbo Hu, Yibo Zhu, Xuanzhe Liu, Xin Jin, and Hao Zhang. 2024. DistServe: Disaggregating prefill and decoding for goodput-optimized large language model serving. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24). 193\u2013210."}],"event":{"name":"ICS '26: 2026 International Conference on Supercomputing","location":"Belfast United Kingdom","acronym":"ICS '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 40th ACM International Conference on Supercomputing"],"original-title":[],"deposited":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T12:47:12Z","timestamp":1782996432000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3797905.3815788"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":43,"alternative-id":["10.1145\/3797905.3815788","10.1145\/3797905"],"URL":"https:\/\/doi.org\/10.1145\/3797905.3815788","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}