{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T17:57:04Z","timestamp":1785952624177,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":84,"publisher":"ACM","funder":[{"name":"University funds","award":[""],"award-info":[{"award-number":[""]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,18]]},"DOI":"10.1145\/3725843.3756111","type":"proceedings-article","created":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T17:19:56Z","timestamp":1760721596000},"page":"626-642","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["Characterizing the Efficiency of Distributed Training: A Power, Performance, and Thermal Perspective"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-9225-7728","authenticated-orcid":false,"given":"Seokjin","family":"Go","sequence":"first","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-8406-0817","authenticated-orcid":false,"given":"Joongun","family":"Park","sequence":"additional","affiliation":[{"name":"Georgia Tech, Atlanta, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5903-2847","authenticated-orcid":false,"given":"Spandan","family":"More","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3718-5272","authenticated-orcid":false,"given":"Hanjiang","family":"Wu","sequence":"additional","affiliation":[{"name":"Georgia Tech, Atlanta, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1912-5834","authenticated-orcid":false,"given":"Irene","family":"Wang","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4302-4227","authenticated-orcid":false,"given":"Aaron","family":"Jezghani","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5738-6942","authenticated-orcid":false,"given":"Tushar","family":"Krishna","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8184-0528","authenticated-orcid":false,"given":"Divya","family":"Mahajan","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,17]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Mart\u00edn Abadi Ashish Agarwal Paul Barham Eugene Brevdo Zhifeng Chen Craig Citro Greg Corrado Andy Davis Jeffrey Dean Matthieu Devin Sanjay Ghemawat Ian Goodfellow Andrew Harp Geoffrey Irving Michael Isard Yangqing Jia Rafal Jozefowicz Lukasz Kaiser Manjunath Kudlur Josh Levenberg Dan Man\u00e9 Rajat Monga Sherry Moore Derek Murray Chris Olah Mike Schuster Jonathon Shlens Benoit Steiner Ilya Sutskever Kunal Talwar Paul Tucker Vincent Vanhoucke Vijay Vasudevan Fernanda Vi\u00e9gas Oriol Vinyals Pete Warden Martin Wattenberg Martin Wicke Yuan Yu and Xiaoqiang Zheng. 2015. TensorFlow: Large-Scale Machine Learning on Heterogeneous Distributed Systems. http:\/\/download.tensorflow.org\/paper\/whitepaper2015.pdf"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","unstructured":"Muhammad Adnan Yassaman\u00a0Ebrahimzadeh Maboud Divya Mahajan and Prashant\u00a0J. Nair. 2021. Accelerating recommendation system training by leveraging popular choices. Proc. VLDB Endow. 15 1 (Sept. 2021) 127\u2013140. 10.14778\/3485450.3485462","DOI":"10.14778\/3485450.3485462"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00081"},{"key":"e_1_3_3_1_5_2","unstructured":"Muhammad Adnan Amar Phanishayee Janardhan Kulkarni Prashant\u00a0J. Nair and Divya Mahajan. 2024. Workload-Aware Hardware Accelerator Mining for Distributed Deep Learning Training. arxiv:https:\/\/arXiv.org\/abs\/2404.14632\u00a0[cs.AR] https:\/\/arxiv.org\/abs\/2404.14632"},{"key":"e_1_3_3_1_6_2","volume-title":"AMD System Management Interface (SMI)","author":"Inc. Advanced Micro Devices,","year":"2024","unstructured":"Advanced Micro Devices, Inc.2024. AMD System Management Interface (SMI). https:\/\/github.com\/RadeonOpenCompute\/rocm_smi_lib."},{"key":"e_1_3_3_1_7_2","unstructured":"Lightning AI. 2025. PyTorch Lightning Docs: 2D Paralleism. https:\/\/lightning.ai\/docs\/pytorch\/stable\/advanced\/model_parallel\/tp_fsdp.html."},{"key":"e_1_3_3_1_8_2","unstructured":"Anthropic. 2024. The Claude 3 Model Family: Opus Sonnet Haiku. https:\/\/api.semanticscholar.org\/CorpusID:268232499"},{"key":"e_1_3_3_1_9_2","series-title":"(ICML\u201923)","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Biderman Stella","year":"2023","unstructured":"Stella Biderman, Hailey Schoelkopf, Quentin Anthony, Herbie Bradley, Kyle O\u2019Brien, Eric Hallahan, Mohammad\u00a0Aflah Khan, Shivanshu Purohit, USVSN\u00a0Sai Prashanth, Edward Raff, Aviya Skowron, Lintang Sutawika, and Oskar Van Der\u00a0Wal. 2023. Pythia: a suite for analyzing large language models across training and scaling. In Proceedings of the 40th International Conference on Machine Learning (Honolulu, Hawaii, USA) (ICML\u201923). JMLR.org, Article 102, 34\u00a0pages."},{"key":"e_1_3_3_1_10_2","unstructured":"Tom\u00a0B. Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell Sandhini Agarwal Ariel Herbert-Voss Gretchen Krueger Tom Henighan Rewon Child Aditya Ramesh Daniel\u00a0M. Ziegler Jeffrey Wu Clemens Winter Christopher Hesse Mark Chen Eric Sigler Mateusz Litwin Scott Gray Benjamin Chess Jack Clark Christopher Berner Sam McCandlish Alec Radford Ilya Sutskever and Dario Amodei. 2020. Language Models are Few-Shot Learners. arxiv:https:\/\/arXiv.org\/abs\/2005.14165\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2005.14165"},{"key":"e_1_3_3_1_11_2","volume-title":"PPoPP 2021","author":"Cai Zixian","year":"2021","unstructured":"Zixian Cai, Zhengyang Liu, Saeed Maleki, Madan Musuvathi, Todd Mytkowicz, Jacob Nelson, and Olli Saarikivi. 2021. Synthesizing optimal collective communication algorithms. In PPoPP 2021. https:\/\/www.microsoft.com\/en-us\/research\/publication\/synthesizing-optimal-collective-communication-algorithms\/"},{"key":"e_1_3_3_1_12_2","unstructured":"Gheorghe Comanici Eric Bieber Mike Schaekermann Ice Pasupat Noveen Sachdeva Inderjit Dhillon Marcel Blistein Ori Ram Dan Zhang Evan Rosen et\u00a0al. 2025. Gemini 2.5: Pushing the frontier with advanced reasoning multimodality long context and next generation agentic capabilities. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2507.06261 (2025)."},{"key":"e_1_3_3_1_13_2","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Dao Tri","year":"2022","unstructured":"Tri Dao, Daniel\u00a0Y. Fu, Stefano Ermon, Atri Rudra, and Christopher R\u00e9. 2022. FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_1_14_2","unstructured":"Forbes. 2025. Graduating The New AI-Ready Workforce: How GA Tech Is Leading The Way. https:\/\/www.forbes.com\/sites\/committeeof200\/2025\/02\/10\/graduating-the-new-ai-ready-workforce-how-ga-tech-is-leading-the-way\/."},{"key":"e_1_3_3_1_15_2","unstructured":"Leo Gao Stella Biderman Sid Black Laurence Golding Travis Hoppe Charles Foster Jason Phang Horace He Anish Thite Noa Nabeshima Shawn Presser and Connor Leahy. 2020. The Pile: An 800GB Dataset of Diverse Text for Language Modeling. arxiv:https:\/\/arXiv.org\/abs\/2101.00027\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2101.00027"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589082"},{"key":"e_1_3_3_1_17_2","unstructured":"Google. 2025. Balance of power: A full-stack approach to power and thermal fluctuations in ML infrastructure. https:\/\/cloud.google.com\/blog\/topics\/systems\/mitigating-power-and-thermal-fluctuations-in-ml-infrastructure."},{"key":"e_1_3_3_1_18_2","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et\u00a0al. 2024. The llama 3 herd of models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.21783 (2024)."},{"key":"e_1_3_3_1_19_2","unstructured":"Aaron Harlap Deepak Narayanan Amar Phanishayee Vivek Seshadri Nikhil Devanur Greg Ganger and Phil Gibbons. 2018. PipeDream: Fast and Efficient Pipeline Parallel DNN Training. arxiv:https:\/\/arXiv.org\/abs\/1806.03377\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/1806.03377"},{"key":"e_1_3_3_1_20_2","unstructured":"Edward\u00a0J. Hu Yelong Shen Phillip Wallis Zeyuan Allen-Zhu Yuanzhi Li Shean Wang Lu Wang and Weizhu Chen. 2021. LoRA: Low-Rank Adaptation of Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2106.09685\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2106.09685"},{"key":"e_1_3_3_1_21_2","series-title":"(NSDI\u201924)","volume-title":"Proceedings of the 21st USENIX Symposium on Networked Systems Design and Implementation","author":"Hu Qinghao","year":"2024","unstructured":"Qinghao Hu, Zhisheng Ye, Zerui Wang, Guoteng Wang, Meng Zhang, Qiaoling Chen, Peng Sun, Dahua Lin, Xiaolin Wang, Yingwei Luo, Yonggang Wen, and Tianwei Zhang. 2024. Characterization of large language model development in the datacenter. In Proceedings of the 21st USENIX Symposium on Networked Systems Design and Implementation (Santa Clara, CA, USA) (NSDI\u201924). USENIX Association, USA, Article 39, 21\u00a0pages."},{"key":"e_1_3_3_1_22_2","unstructured":"Yanping Huang Youlong Cheng Ankur Bapna Orhan Firat Mia\u00a0Xu Chen Dehao Chen HyoukJoong Lee Jiquan Ngiam Quoc\u00a0V. Le Yonghui Wu and Zhifeng Chen. 2019. GPipe: Efficient Training of Giant Neural Networks using Pipeline Parallelism. arxiv:https:\/\/arXiv.org\/abs\/1811.06965\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1811.06965"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"crossref","unstructured":"Sagar Imambi Kolla\u00a0Bhanu Prakash and GR Kanagachidambaresan. 2021. PyTorch. Programming with TensorFlow: solution for edge computing applications (2021) 87\u2013104.","DOI":"10.1007\/978-3-030-57077-4_10"},{"key":"e_1_3_3_1_24_2","unstructured":"Zhihao Jia Matei Zaharia and Alex Aiken. 2018. Beyond Data and Model Parallelism for Deep Neural Networks. arxiv:https:\/\/arXiv.org\/abs\/1807.05358\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/1807.05358"},{"key":"e_1_3_3_1_25_2","unstructured":"Albert\u00a0Q Jiang Alexandre Sablayrolles Antoine Roux Arthur Mensch Blanche Savary Chris Bamford Devendra\u00a0Singh Chaplot Diego de\u00a0las Casas Emma\u00a0Bou Hanna Florian Bressand et\u00a0al. 2024. Mixtral of experts. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.04088 (2024)."},{"key":"e_1_3_3_1_26_2","unstructured":"Ziheng Jiang Haibin Lin Yinmin Zhong Qi Huang Yangrui Chen Zhi Zhang Yanghua Peng Xiang Li Cong Xie Shibiao Nong Yulu Jia Sun He Hongmin Chen Zhihao Bai Qi Hou Shipeng Yan Ding Zhou Yiyao Sheng Zhuo Jiang Haohan Xu Haoran Wei Zhang Zhang Pengfei Nie Leqi Zou Sida Zhao Liang Xiang Zherui Liu Zhe Li Xiaoying Jia Jianxi Ye Xin Jin and Xin Liu. 2024. MegaScale: Scaling Large Language Model Training to More Than 10 000 GPUs. arxiv:https:\/\/arXiv.org\/abs\/2402.15627\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2402.15627"},{"key":"e_1_3_3_1_27_2","unstructured":"Qiao Jin Bhuwan Dhingra Zhengping Liu William\u00a0W. Cohen and Xinghua Lu. 2019. PubMedQA: A Dataset for Biomedical Research Question Answering. arxiv:https:\/\/arXiv.org\/abs\/1909.06146\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/1909.06146"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO61859.2024.00070"},{"key":"e_1_3_3_1_29_2","unstructured":"Vijay Korthikanti Jared Casper Sangkug Lym Lawrence McAfee Michael Andersch Mohammad Shoeybi and Bryan Catanzaro. 2022. Reducing Activation Recomputation in Large Transformer Models. arxiv:https:\/\/arXiv.org\/abs\/2205.05198\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2205.05198"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"crossref","unstructured":"Alex Krizhevsky Ilya Sutskever and Geoffrey\u00a0E. Hinton. 2017. ImageNet Classification with Deep Convolutional Neural Networks. Commun. ACM 60 6 (May 2017) 84\u201390.","DOI":"10.1145\/3065386"},{"key":"e_1_3_3_1_31_2","unstructured":"Oleksii Kuchaiev Jason Li Huyen Nguyen Oleksii Hrinchuk Ryan Leary Boris Ginsburg Samuel Kriman Stanislav Beliaev Vitaly Lavrukhin Jack Cook et\u00a0al. 2019. Nemo: a toolkit for building ai applications using neural modules. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1909.09577 (2019)."},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS64960.2025.00041"},{"key":"e_1_3_3_1_33_2","unstructured":"Seonho Lee Jihwan Oh Junkyum Kim Seokjin Go Jongse Park and Divya Mahajan. 2025. Characterizing Compute-Communication Overlap in GPU-Accelerated Distributed Deep Learning: Performance and Power Implications. arxiv:https:\/\/arXiv.org\/abs\/2507.03114\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2507.03114"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2018.8573483"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","unstructured":"Shen Li Yanli Zhao Rohan Varma Omkar Salpekar Pieter Noordhuis Teng Li Adam Paszke Jeff Smith Brian Vaughan Pritam Damania and Soumith Chintala. 2020. PyTorch distributed: experiences on accelerating data parallel training. Proc. VLDB Endow. 13 12 (Aug. 2020) 3005\u20133018. 10.14778\/3415478.3415530","DOI":"10.14778\/3415478.3415530"},{"key":"e_1_3_3_1_36_2","unstructured":"Aixin Liu Bei Feng Bing Xue Bingxuan Wang Bochao Wu Chengda Lu Chenggang Zhao Chengqi Deng Chenyu Zhang Chong Ruan et\u00a0al. 2024. Deepseek-v3 technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.19437 (2024)."},{"key":"e_1_3_3_1_37_2","unstructured":"Divya Mahajan. 2024. Training a machine learning model using an acceleration pipeline with popular and non-popular micro-batches. https:\/\/patents.google.com\/patent\/US20240320054A1\/en US Patent App. 18\/126 350."},{"key":"e_1_3_3_1_38_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2016.7446050"},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"publisher","unstructured":"Xinxin Mei and Xiaowen Chu. 2017. Dissecting GPU Memory Hierarchy Through Microbenchmarking. IEEE Trans. Parallel Distrib. Syst. 28 1 (Jan. 2017) 72\u201386. 10.1109\/TPDS.2016.2549523","DOI":"10.1109\/TPDS.2016.2549523"},{"key":"e_1_3_3_1_40_2","unstructured":"Meta. 2024. 2024 Sustainability Report. https:\/\/sustainability.atmeta.com\/2024-sustainability-report\/."},{"key":"e_1_3_3_1_41_2","unstructured":"Deepak Narayanan Amar Phanishayee Kaiyu Shi Xie Chen and Matei Zaharia. 2021. Memory-Efficient Pipeline-Parallel DNN Training."},{"key":"e_1_3_3_1_42_2","doi-asserted-by":"crossref","unstructured":"Deepak Narayanan Mohammad Shoeybi Jared Casper Patrick LeGresley Mostofa Patwary Vijay\u00a0Anand Korthikanti Dmitri Vainbrand Prethvi Kashinkunti Julie Bernauer Bryan Catanzaro Amar Phanishayee and Matei Zaharia. 2021. Efficient Large-Scale Language Model Training on GPU Clusters Using Megatron-LM. arxiv:https:\/\/arXiv.org\/abs\/2104.04473\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2104.04473","DOI":"10.1145\/3458817.3476209"},{"key":"e_1_3_3_1_43_2","unstructured":"Nvidia. 2024. NVIDIA DGX Cooling and Airflow Optimization. https:\/\/docs.nvidia.com\/dgx-superpod\/design-guides\/dgx-superpod-data-center-design-h100\/latest\/cooling.html."},{"key":"e_1_3_3_1_44_2","unstructured":"Nvidia. 2025. InfiniBand. https:\/\/www.nvidia.com\/en-us\/networking\/products\/infiniband\/."},{"key":"e_1_3_3_1_45_2","unstructured":"Nvidia. 2025. NVIDIA Megatron Core Documentation. https:\/\/github.com\/NVIDIA\/Megatron-LM\/tree\/main\/megatron\/core\/transformer\/moe."},{"key":"e_1_3_3_1_46_2","unstructured":"Nvidia. 2025. NVIDIA NeMo Documentation: Parallelism. https:\/\/docs.nvidia.com\/nemo-framework\/user-guide\/latest\/nemotoolkit\/features\/parallelisms.html."},{"key":"e_1_3_3_1_47_2","unstructured":"Nvidia. 2025. NVlink. https:\/\/www.nvidia.com\/en-us\/data-center\/nvlink\/."},{"key":"e_1_3_3_1_48_2","volume-title":"NVIDIA Management Library (NVML)","author":"Corporation NVIDIA","year":"2024","unstructured":"NVIDIA Corporation. 2024. NVIDIA Management Library (NVML). https:\/\/docs.nvidia.com\/deploy\/nvml-api\/index.html."},{"key":"e_1_3_3_1_49_2","unstructured":"Georgia\u00a0Institute of Technology. 2024. Georgia Tech AI Makerspace. https:\/\/coe.gatech.edu\/academics\/ai-for-engineering\/ai-makerspace."},{"key":"e_1_3_3_1_50_2","doi-asserted-by":"publisher","DOI":"10.1145\/3123939.3123979"},{"key":"e_1_3_3_1_51_2","volume-title":"NIPS-W","author":"Paszke Adam","year":"2017","unstructured":"Adam Paszke, Sam Gross, Soumith Chintala, Gregory Chanan, Edward Yang, Zachary DeVito, Zeming Lin, Alban Desmaison, Luca Antiga, and Adam Lerer. 2017. Automatic differentiation in PyTorch. In NIPS-W."},{"key":"e_1_3_3_1_52_2","unstructured":"Pratyush Patel Esha Choukse Chaojie Zhang Aashaka Shah \u00cd\u00f1igo Goiri Saeed Maleki and Ricardo Bianchini. 2024. Splitwise: Efficient generative LLM inference using phase splitting. arxiv:https:\/\/arXiv.org\/abs\/2311.18677\u00a0[cs.AR] https:\/\/arxiv.org\/abs\/2311.18677"},{"key":"e_1_3_3_1_53_2","unstructured":"Amar Phanishayee Divya Mahajan Janardhan Kulkarni Miguel Castro and Muhammad Adnan. 2022. Workload-Aware Hardware Architecture Recommendations. https:\/\/patents.google.com\/patent\/US20240126611A1\/en US Patent App. 17\/965 681."},{"key":"e_1_3_3_1_54_2","unstructured":"Amar Phanishayee Divya Mahajan and Jakub\u00a0Michal Tarnawski. 2025. Integrated Hardware Architecture and Distribution Strategy Optimization for Deep Learning Models. https:\/\/patents.google.com\/patent\/US20250061533A1\/en US Patent App. 18\/452 162."},{"key":"e_1_3_3_1_55_2","first-page":"606","volume-title":"Proceedings of Machine Learning and Systems","volume":"5","author":"Pope Reiner","year":"2023","unstructured":"Reiner Pope, Sholto Douglas, Aakanksha Chowdhery, Jacob Devlin, James Bradbury, Jonathan Heek, Kefan Xiao, Shivani Agrawal, and Jeff Dean. 2023. Efficiently Scaling Transformer Inference. In Proceedings of Machine Learning and Systems , D.\u00a0Song, M.\u00a0Carbin, and T.\u00a0Chen (Eds.), Vol.\u00a05. Curan, 606\u2013624. https:\/\/proceedings.mlsys.org\/paper_files\/paper\/2023\/file\/c4be71ab8d24cdfb45e3d06dbfca2780-Paper-mlsys2023.pdf"},{"key":"e_1_3_3_1_56_2","unstructured":"PyTorch. 2025. Large Scale Transformer model training with Tensor Parallel (TP). https:\/\/docs.pytorch.org\/tutorials\/intermediate\/TP_tutorial.html."},{"key":"e_1_3_3_1_57_2","unstructured":"Samyam Rajbhandari Conglong Li Zhewei Yao Minjia Zhang Reza\u00a0Yazdani Aminabadi Ammar\u00a0Ahmad Awan Jeff Rasley and Yuxiong He. 2022. DeepSpeed-MoE: Advancing Mixture-of-Experts Inference and Training to Power Next-Generation AI Scale. arxiv:https:\/\/arXiv.org\/abs\/2201.05596\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2201.05596"},{"key":"e_1_3_3_1_58_2","unstructured":"Samyam Rajbhandari Jeff Rasley Olatunji Ruwase and Yuxiong He. 2020. ZeRO: Memory Optimizations Toward Training Trillion Parameter Models. arxiv:https:\/\/arXiv.org\/abs\/1910.02054\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/1910.02054"},{"key":"e_1_3_3_1_59_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS48437.2020.00018"},{"key":"e_1_3_3_1_60_2","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406703"},{"key":"e_1_3_3_1_61_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613155"},{"key":"e_1_3_3_1_62_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613156"},{"key":"e_1_3_3_1_63_2","first-page":"593","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Shah Aashaka","year":"2023","unstructured":"Aashaka Shah, Vijay Chidambaram, Meghan Cowan, Saeed Maleki, Madan Musuvathi, Todd Mytkowicz, Jacob Nelson, Olli Saarikivi, and Rachee Singh. 2023. TACCL: Guiding Collective Algorithm Synthesis using Communication Sketches. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23). USENIX Association, Boston, MA, 593\u2013612. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/shah"},{"key":"e_1_3_3_1_64_2","doi-asserted-by":"crossref","unstructured":"Hardik Sharma Jongse Park Divya Mahajan Emmanuel Amaro Joon\u00a0Kyung Kim Chenkai Shao Asit Mishra and Hadi Esmaeilzadeh. 2016. From High-Level Deep Neural Models to FPGAs.","DOI":"10.1109\/MICRO.2016.7783720"},{"key":"e_1_3_3_1_65_2","unstructured":"Mohammad Shoeybi Mostofa Patwary Raul Puri Patrick LeGresley Jared Casper and Bryan Catanzaro. 2019. Megatron-lm: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1909.08053 (2019)."},{"key":"e_1_3_3_1_66_2","doi-asserted-by":"publisher","DOI":"10.1145\/3577193.3593704"},{"key":"e_1_3_3_1_67_2","unstructured":"Siddharth Singh Prajwal Singhania Aditya Ranjan John Kirchenbauer Jonas Geiping Yuxin Wen Neel Jain Abhimanyu Hans Manli Shu Aditya Tomar Tom Goldstein and Abhinav Bhatele. 2025. Democratizing AI: Open-source Scalable LLM Training on GPU-based Supercomputers. arxiv:https:\/\/arXiv.org\/abs\/2502.08145\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2502.08145"},{"key":"e_1_3_3_1_68_2","unstructured":"Srinivas Sridharan Taekyung Heo Louis Feng Zhaodong Wang Matt Bergeron Wenyin Fu Shengbao Zheng Brian Coutinho Saeed Rashidi Changhai Man and Tushar Krishna. 2023. Chakra: Advancing Performance Benchmarking and Co-design using Standardized Execution Traces. arxiv:https:\/\/arXiv.org\/abs\/2305.14516\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2305.14516"},{"key":"e_1_3_3_1_69_2","unstructured":"Jovan Stojkovic Chaojie Zhang \u00cd\u00f1igo Goiri Esha Choukse Haoran Qiu Rodrigo Fonseca Josep Torrellas and Ricardo Bianchini. 2025. TAPAS: Thermal- and Power-Aware Scheduling for LLM Inference in Cloud Platforms. arxiv:https:\/\/arXiv.org\/abs\/2501.02600\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2501.02600"},{"key":"e_1_3_3_1_70_2","unstructured":"Jovan Stojkovic Chaojie Zhang \u00cd\u00f1igo Goiri Josep Torrellas and Esha Choukse. 2024. DynamoLLM: Designing LLM Inference Clusters for Performance and Energy Efficiency. arxiv:https:\/\/arXiv.org\/abs\/2408.00741\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2408.00741"},{"key":"e_1_3_3_1_71_2","series-title":"(NIPS \u201921)","volume-title":"Proceedings of the 35th International Conference on Neural Information Processing Systems","author":"Tarnawski Jakub","year":"2021","unstructured":"Jakub Tarnawski, Deepak Narayanan, and Amar Phanishayee. 2021. Piper: multidimensional planner for DNN parallelization. In Proceedings of the 35th International Conference on Neural Information Processing Systems(NIPS \u201921). Curran Associates Inc., Red Hook, NY, USA, Article 1902, 12\u00a0pages."},{"key":"e_1_3_3_1_72_2","unstructured":"Jakub\u00a0M Tarnawski Amar Phanishayee Nikhil Devanur Divya Mahajan and Fanny Nina\u00a0Paravecino. 2020. Efficient algorithms for device placement of dnn graph operators. Advances in Neural Information Processing Systems 33 (2020)."},{"key":"e_1_3_3_1_73_2","unstructured":"Guanhua Wang S. Venkataraman Amar Phanishayee Jorgen Thelin Nikhil\u00a0R. Devanur and I. Stoica. 2020. Blink: Fast and Generic Collectives for Distributed ML. (2020) 1\u201315."},{"key":"e_1_3_3_1_74_2","volume-title":"Thirty-seventh Conference on Neural Information Processing Systems","author":"Wang Irene","year":"2023","unstructured":"Irene Wang, Prashant\u00a0J. Nair, and Divya Mahajan. 2023. FLuID: Mitigating Stragglers in Federated Learning using Invariant Dropout. In Thirty-seventh Conference on Neural Information Processing Systems."},{"key":"e_1_3_3_1_75_2","unstructured":"Irene Wang Jakub Tarnawski Amar Phanishayee and Divya Mahajan. 2024. Integrated Hardware Architecture and Device Placement Search. arxiv:https:\/\/arXiv.org\/abs\/2407.13143\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2407.13143"},{"key":"e_1_3_3_1_76_2","doi-asserted-by":"publisher","DOI":"10.1145\/3567955.3567959"},{"key":"e_1_3_3_1_77_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613149"},{"key":"e_1_3_3_1_78_2","first-page":"739","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Wang Weiyang","year":"2023","unstructured":"Weiyang Wang, Moein Khazraee, Zhizhen Zhong, Manya Ghobadi, Zhihao Jia, Dheevatsa Mudigere, Ying Zhang, and Anthony Kewitsch. 2023. TopoOpt: Co-optimizing Network Topology and Parallelization Strategy for Distributed Training Jobs. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23). USENIX Association, Boston, MA, 739\u2013767. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/wang-weiyang"},{"key":"e_1_3_3_1_79_2","unstructured":"William Won Midhilesh Elavazhagan Sudarshan Srinivasan Swati Gupta and Tushar Krishna. 2024. TACOS: Topology-Aware Collective Algorithm Synthesizer for Distributed Machine Learning. arxiv:https:\/\/arXiv.org\/abs\/2304.05301\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2304.05301"},{"key":"e_1_3_3_1_80_2","doi-asserted-by":"publisher","DOI":"10.1109\/ispass57527.2023.00035"},{"key":"e_1_3_3_1_81_2","first-page":"119","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"You Jie","year":"2023","unstructured":"Jie You, Jae-Won Chung, and Mosharaf Chowdhury. 2023. Zeus: Understanding and optimizing { GPU} energy consumption of { DNN} training. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23). 119\u2013139."},{"key":"e_1_3_3_1_82_2","unstructured":"Ruisi Zhang Tianyu Liu Will Feng Andrew Gu Sanket Purandare Wanchao Liang and Francisco Massa. 2024. SimpleFSDP: Simpler Fully Sharded Data Parallel with torch.compile. arxiv:https:\/\/arXiv.org\/abs\/2411.00284\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2411.00284"},{"key":"e_1_3_3_1_83_2","unstructured":"Yanli Zhao Andrew Gu Rohan Varma Liang Luo Chien-Chin Huang Min Xu Less Wright Hamid Shojanazeri Myle Ott Sam Shleifer Alban Desmaison Can Balioglu Pritam Damania Bernard Nguyen Geeta Chauhan Yuchen Hao Ajit Mathews and Shen Li. 2023. PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel. arxiv:https:\/\/arXiv.org\/abs\/2304.11277\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2304.11277"},{"key":"e_1_3_3_1_84_2","unstructured":"Lianmin Zheng Zhuohan Li Hao Zhang Yonghao Zhuang Zhifeng Chen Yanping Huang Yida Wang Yuanzhong Xu Danyang Zhuo Eric\u00a0P. Xing Joseph\u00a0E. Gonzalez and Ion Stoica. 2022. Alpa: Automating Inter- and Intra-Operator Parallelism for Distributed Deep Learning. arxiv:https:\/\/arXiv.org\/abs\/2201.12023\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2201.12023"},{"key":"e_1_3_3_1_85_2","unstructured":"Ruidong Zhu Ziheng Jiang Chao Jin Peng Wu Cesar\u00a0A. Stuardo Dongyang Wang Xinlei Zhang Huaping Zhou Haoran Wei Yang Cheng Jianzhe Xiao Xinyi Zhang Lingjun Liu Haibin Lin Li-Wen Chang Jianxi Ye Xiao Yu Xuanzhe Liu Xin Jin and Xin Liu. 2025. MegaScale-Infer: Serving Mixture-of-Experts at Scale with Disaggregated Expert Parallelism. arxiv:https:\/\/arXiv.org\/abs\/2504.02263\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2504.02263"}],"event":{"name":"MICRO 2025: 58th IEEE\/ACM International Symposium on Microarchitecture","location":"Seoul Korea","acronym":"MICRO 2025","sponsor":["SIGMICRO ACM Special Interest Group on Microarchitectural Research and Processing"]},"container-title":["Proceedings of the 58th IEEE\/ACM International Symposium on Microarchitecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3725843.3756111","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,26]],"date-time":"2026-01-26T21:42:43Z","timestamp":1769463763000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3725843.3756111"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,17]]},"references-count":84,"alternative-id":["10.1145\/3725843.3756111","10.1145\/3725843"],"URL":"https:\/\/doi.org\/10.1145\/3725843.3756111","relation":{},"subject":[],"published":{"date-parts":[[2025,10,17]]},"assertion":[{"value":"2025-10-17","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}