{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:28:57Z","timestamp":1784136537551,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":84,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,11,13]],"date-time":"2021-11-13T00:00:00Z","timestamp":1636761600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,11,14]]},"DOI":"10.1145\/3458817.3476223","type":"proceedings-article","created":{"date-parts":[[2021,10,21]],"date-time":"2021-10-21T05:10:34Z","timestamp":1634793034000},"page":"1-15","source":"Crossref","is-referenced-by-count":127,"title":["Characterization and prediction of deep learning workloads in large-scale GPU datacenters"],"prefix":"10.1145","author":[{"given":"Qinghao","family":"Hu","sequence":"first","affiliation":[{"name":"Nanyang Technological University and Nanyang Technological University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Sun","sequence":"additional","affiliation":[{"name":"SenseTime"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shengen","family":"Yan","sequence":"additional","affiliation":[{"name":"SenseTime"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yonggang","family":"Wen","sequence":"additional","affiliation":[{"name":"Nanyang Technological University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tianwei","family":"Zhang","sequence":"additional","affiliation":[{"name":"Nanyang Technological University"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2021,11,13]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"2021. DGX-1 BMC. https:\/\/docs.nvidia.com\/dgx\/dgx1-user-guide.  2021. DGX-1 BMC. https:\/\/docs.nvidia.com\/dgx\/dgx1-user-guide."},{"key":"e_1_3_2_2_2_1","unstructured":"2021. Lustre. https:\/\/www.lustre.org\/.  2021. Lustre. https:\/\/www.lustre.org\/."},{"key":"e_1_3_2_2_3_1","unstructured":"2021. Memcached. https:\/\/memcached.org\/.  2021. Memcached. https:\/\/memcached.org\/."},{"key":"e_1_3_2_2_4_1","unstructured":"2021. NCCL. https:\/\/developer.nvidia.com\/nccl.  2021. NCCL. https:\/\/developer.nvidia.com\/nccl."},{"key":"e_1_3_2_2_5_1","unstructured":"2021. NVIDIA Multi-Instance GPU. https:\/\/www.nvidia.com\/en-us\/technologies\/multi-instance-gpu\/.  2021. NVIDIA Multi-Instance GPU. https:\/\/www.nvidia.com\/en-us\/technologies\/multi-instance-gpu\/."},{"key":"e_1_3_2_2_6_1","unstructured":"2021. NVIDIA-smi. https:\/\/developer.nvidia.com\/nvidia-system-management-interface.  2021. NVIDIA-smi. https:\/\/developer.nvidia.com\/nvidia-system-management-interface."},{"key":"e_1_3_2_2_7_1","unstructured":"2021. NVLink. https:\/\/www.nvidia.com\/en-us\/data-center\/nvlink\/.  2021. NVLink. https:\/\/www.nvidia.com\/en-us\/data-center\/nvlink\/."},{"key":"e_1_3_2_2_8_1","volume-title":"Power and Performance Characterization and Modeling of GPU-Accelerated Systems. In 2014 IEEE 28th International Parallel and Distributed Processing Symposium (IPDPS '14)","author":"Abe Yuki","year":"2014"},{"key":"e_1_3_2_2_9_1","volume-title":"2018 USENIX Annual Technical Conference (USENIX ATC '18)","author":"Amvrosiadis George","year":"2018"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"crossref","first-page":"2041","DOI":"10.1016\/j.cpc.2010.11.011","article-title":"Trends in supercomputing: The European path to exascale","volume":"182","author":"Attig Norbert","year":"2011","journal-title":"Computer Physics Communications"},{"key":"e_1_3_2_2_11_1","volume-title":"3rd International Conference on Learning Representations (ICLR '15)","author":"Bahdanau Dzmitry","year":"2015"},{"key":"e_1_3_2_2_12_1","volume-title":"Proceedings of the 26th ACM International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS '21)","author":"Bl\u00f6cher Marcel","year":"2021"},{"key":"e_1_3_2_2_13_1","volume-title":"Apollo: Scalable and Coordinated Scheduling for Cloud-Scale Computing. In 11th USENIX Symposium on Operating Systems Design and Implementation (OSDI '14)","author":"Boutin Eric","year":"2014"},{"key":"e_1_3_2_2_14_1","volume-title":"Mintz","author":"Bridges Robert A.","year":"2016"},{"key":"e_1_3_2_2_15_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS '20)","author":"Brown Tom","year":"2020"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"crossref","first-page":"70","DOI":"10.1145\/2898442.2898444","article-title":"Borg, Omega, and Kubernetes: Lessons Learned from Three ContainerManagement Systems over a Decade","volume":"14","author":"Burns Brendan","year":"2016","journal-title":"Queue"},{"key":"e_1_3_2_2_17_1","volume-title":"Proceedings of the Eighth International Conference on Future Energy Systems (e-Energy '17)","author":"Chau Vincent","year":"2017"},{"key":"e_1_3_2_2_18_1","volume-title":"Proceedings of the Fifteenth European Conference on Computer Systems (EuroSys '20)","author":"Chaudhary Shubham","year":"2020"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"crossref","first-page":"1802","DOI":"10.14778\/2367502.2367519","article-title":"Interactive Analytical Processing in Big Data Systems: A Cross-Industry Study of MapReduce Workloads","volume":"5","author":"Chen Yanpei","year":"2012","journal-title":"Proceedings of the VLDB Endowment"},{"key":"e_1_3_2_2_20_1","volume-title":"Proceedings of the 26th Symposium on Operating Systems Principles (SOSP '17)","author":"Cortez Eli","year":"2017"},{"key":"e_1_3_2_2_21_1","volume-title":"Proceedings of the 10th ACM Conference on Recommender Systems (RecSys '16)","author":"Covington Paul","year":"2016"},{"key":"e_1_3_2_2_22_1","volume-title":"Proceedings of the ACM Symposium on Cloud Computing (SoCC '14)","author":"Curino Carlo","year":"2014"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"crossref","first-page":"732","DOI":"10.1109\/COMST.2015.2481183","article-title":"Data center energy consumption modeling: A survey","volume":"18","author":"Dayarathna Miyuru","year":"2016","journal-title":"IEEE Communications Surveys and Tutorials"},{"key":"e_1_3_2_2_24_1","volume-title":"Hawk: Hybrid Datacenter Scheduling. In 2015 USENIX Annual Technical Conference (USENIX ATC '15)","author":"Delgado Pamela","year":"2015"},{"key":"e_1_3_2_2_25_1","volume-title":"Switch Transformers: Scaling to Trillion Parameter Models with Simple and Efficient Sparsity. CoRR abs\/2101.03961","author":"Fedus William","year":"2021"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"crossref","unstructured":"Dror G. Feitelson. 1996. Packing schemes for gang scheduling. In Job Scheduling Strategies for Parallel Processing.  Dror G. Feitelson. 1996. Packing schemes for gang scheduling. In Job Scheduling Strategies for Parallel Processing.","DOI":"10.1007\/BFb0022283"},{"key":"e_1_3_2_2_27_1","volume-title":"Proceedings of the 7th ACM European Conference on Computer Systems (EuroSys '12)","author":"Ferguson Andrew D.","year":"2012"},{"key":"e_1_3_2_2_28_1","volume-title":"42nd International Conference on Parallel Processing (ICPP '13)","author":"Ge Rong","year":"2013"},{"key":"e_1_3_2_2_29_1","volume-title":"Altruistic Scheduling in Multi-Resource Clusters. In 12th USENIX Symposium on Operating Systems Design and Implementation (OSDI '16)","author":"Grandl Robert","year":"2016"},{"key":"e_1_3_2_2_30_1","volume-title":"GRAPHENE: Packing and Dependency-Aware Scheduling for Data-Parallel Clusters. In 12th USENIX Symposium on Operating Systems Design and Implementation (OSDI '16)","author":"Grandl Robert","year":"2016"},{"key":"e_1_3_2_2_31_1","volume-title":"Tiresias: A GPU Cluster Manager for Distributed Deep Learning. In 16th USENIX Symposium on Networked Systems Design and Implementation (NSDI '19)","author":"Gu Juncheng","year":"2019"},{"key":"e_1_3_2_2_32_1","volume-title":"Time Series Analysis","author":"Hamilton James Douglas"},{"key":"e_1_3_2_2_33_1","volume-title":"Mesos: A Platform for Fine-Grained Resource Sharing in the Data Center. In 8th USENIX Symposium on Networked Systems Design and Implementation (NSDI '11)","author":"Hindman Benjamin","year":"2011"},{"key":"e_1_3_2_2_34_1","volume-title":"Elastic Resource Sharing for Distributed Deep Learning. In 18th USENIX Symposium on Networked Systems Design and Implementation (NSDI '21)","author":"Hwang Changho","year":"2021"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"crossref","first-page":"679","DOI":"10.1016\/j.ijforecast.2006.03.001","article-title":"Another look at measures of forecast accuracy","volume":"22","author":"Hyndman Rob J.","year":"2006","journal-title":"International Journal of Forecasting"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"crossref","first-page":"41","DOI":"10.1145\/1290672.1290678","article-title":"Algorithms for Power Savings","volume":"3","author":"Irani Sandy","year":"2007","journal-title":"ACM Transactions on Algorithms"},{"key":"e_1_3_2_2_37_1","volume-title":"Proceedings of the 2nd ACM SIGOPS\/EuroSys European Conference on Computer Systems (EuroSys '07)","author":"Isard Michael","year":"2007"},{"key":"e_1_3_2_2_38_1","volume-title":"Proceedings of the 2015 ACM Conference on Special Interest Group on Data Communication (SIGCOMM '15)","author":"Jalaparti Virajith","year":"2015"},{"key":"e_1_3_2_2_39_1","volume-title":"Analysis of Large-Scale Multi-Tenant GPU Clusters for DNN Training Workloads. In 2019 USENIX Annual Technical Conference (USENIX ATC '19)","author":"Jeon Myeongjae","year":"2019"},{"key":"e_1_3_2_2_40_1","volume-title":"Proceedings of the 26th ACM International Conference on Supercomputing (ICS '12)","author":"Joubert Wayne","year":"2012"},{"key":"e_1_3_2_2_41_1","volume-title":"Morpheus: Towards Automated SLOs for Enterprise Clusters. In 12th USENIX Symposium on Operating Systems Design and Implementation (OSDI '16)","author":"Jyothi Sangeetha Abdu","year":"2016"},{"key":"e_1_3_2_2_42_1","volume-title":"LightGBM: A Highly Efficient Gradient Boosting Decision Tree. In Advances in Neural Information Processing Systems (NeurIPS '17)","author":"Ke Guolin","year":"2017"},{"key":"e_1_3_2_2_43_1","volume-title":"Proceedings of the Fifteenth European Conference on Computer Systems (EuroSys '20)","author":"Le Tan N.","year":"2020"},{"key":"e_1_3_2_2_44_1","volume-title":"Proceedings of the 11th USENIX Conference on Operating Systems Design and Implementation (OSDI '14)","author":"Li Mu","year":"2014"},{"key":"e_1_3_2_2_45_1","volume-title":"25th IEEE International Symposium on Parallel and Distributed Processing, Workshop Proceedings (IPDPS '11)","author":"Liu Wenjie","year":"2011"},{"key":"e_1_3_2_2_46_1","volume-title":"Themis: Fair and Efficient GPU Cluster Scheduling. In 17th USENIX Symposium on Networked Systems Design and Implementation (NSDI '20)","author":"Mahajan Kshiteej","year":"2020"},{"key":"e_1_3_2_2_47_1","volume-title":"IEEE Conference on Computer Communications (INFOCOM '17)","author":"Mei Xinxin","year":"2017"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"crossref","first-page":"89","DOI":"10.1016\/j.dcan.2016.10.001","article-title":"A survey and measurement study of GPU DVFS on energy conservation","volume":"3","author":"Mei Xinxin","year":"2017","journal-title":"Digital Communications and Networks"},{"key":"e_1_3_2_2_49_1","volume-title":"Energy-aware Task Scheduling with Deadline Constraint in DVFS-enabled Heterogeneous Clusters. CoRR abs\/2104.00486","author":"Mei Xinxin","year":"2021"},{"key":"e_1_3_2_2_50_1","volume-title":"Proceedings of the 14th International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS '09)","author":"Meisner David"},{"key":"e_1_3_2_2_51_1","volume-title":"Proceedings of the 1st Conference of the Extreme Science and Engineering Discovery Environment: Bridging from the EXtreme to the Campus and Beyond (XSEDE '12)","author":"Moore Richard L."},{"key":"e_1_3_2_2_52_1","volume-title":"Heterogeneity-Aware Cluster Scheduling Policies for Deep Learning Workloads. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI '20)","author":"Narayanan Deepak","year":"2020"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"crossref","first-page":"31","DOI":"10.1145\/375360.375365","article-title":"A Guided Tour to Approximate String","volume":"33","author":"Navarro Gonzalo","year":"2001","journal-title":"Matching. Comput. Surveys"},{"key":"e_1_3_2_2_54_1","volume-title":"Proceedings of the Thirteenth EuroSys Conference (EuroSys '18)","author":"Park Jun Woo"},{"key":"e_1_3_2_2_55_1","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis (SC '20)","author":"Patel Tirthak","year":"2020"},{"key":"e_1_3_2_2_56_1","volume-title":"Proceedings of the Thirteenth EuroSys Conference (EuroSys '18)","author":"Peng Yanghua","year":"2018"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"crossref","unstructured":"Lutz Prechelt. 1998. Early Stopping - But When? Springer Berlin Heidelberg 55--69.  Lutz Prechelt. 1998. Early Stopping - But When? Springer Berlin Heidelberg 55--69.","DOI":"10.1007\/3-540-49430-8_3"},{"key":"e_1_3_2_2_58_1","volume-title":"Pollux: Co-adaptive Cluster Scheduling for Goodput-Optimized Deep Learning. In 15th USENIX Symposium on Operating Systems Design and Implementation (OSDI '21)","author":"Qiao Aurick"},{"key":"e_1_3_2_2_59_1","volume-title":"Proceedings of the ACM Symposium on Cloud Computing (SoCC '12)","author":"Reiss Charles"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"crossref","first-page":"206","DOI":"10.1016\/j.jpdc.2017.09.002","article-title":"Towards understanding HPC users and systems: A NERSC case study","volume":"111","author":"Rodrigo Gonzalo P.","year":"2018","journal-title":"J. Parallel and Distrib. Comput."},{"key":"e_1_3_2_2_61_1","volume-title":"2020 USENIX Annual Technical Conference (USENIX ATC '20)","author":"Shahrad Mohammad","year":"2020"},{"key":"e_1_3_2_2_62_1","volume-title":"Furlani","author":"Simakov Nikolay A.","year":"2018"},{"key":"e_1_3_2_2_63_1","volume-title":"E-LAS: Design and Analysis of Completion-Time Agnostic Scheduling for Distributed Deep Learning Cluster. In 49th International Conference on Parallel Processing (ICPP '20)","author":"Sultana Abeda","year":"2020"},{"key":"e_1_3_2_2_64_1","volume-title":"Proceedings of the 27th International Conference on Neural Information Processing Systems (NeurIPS '14)","author":"Sutskever Ilya"},{"key":"e_1_3_2_2_65_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR '14)","author":"Taigman Yaniv","year":"2014"},{"key":"e_1_3_2_2_66_1","volume-title":"Proceedings of the Tenth ACM International Conference on Future Energy Systems (e-Energy '19)","author":"Tang Zhenheng","year":"2019"},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"crossref","first-page":"37","DOI":"10.1080\/00031305.2017.1380080","article-title":"Forecasting at Scale","volume":"72","author":"Taylor Sean J.","year":"2018","journal-title":"The American Statistician"},{"key":"e_1_3_2_2_68_1","volume-title":"Proceedings of the Fifteenth European Conference on Computer Systems (EuroSys '20)","author":"Tirmazi Muhammad","year":"2020"},{"key":"e_1_3_2_2_69_1","volume-title":"Michael A. Kozuch, and Gregory R. Ganger.","author":"Tumanov Alexey","year":"2016"},{"key":"e_1_3_2_2_70_1","volume-title":"Proceedings of the 4th Annual Symposium on Cloud Computing (SoCC '13)","author":"Vavilapalli Vinod Kumar","year":"2013"},{"key":"e_1_3_2_2_71_1","volume-title":"Ernest: Efficient Performance Prediction for Large-Scale Advanced Analytics. In 13th USENIX Symposium on Networked Systems Design and Implementation (NSDI '16)","author":"Venkataraman Shivaram","year":"2016"},{"key":"e_1_3_2_2_72_1","volume-title":"Proceedings of the 8th ACM International Conference on Autonomic Computing (ICAC '11)","author":"Verma Abhishek"},{"key":"e_1_3_2_2_73_1","volume-title":"Proceedings of the 2019 IEEE International Symposium on Workload Characterization (IISWC '19)","author":"Wang Mengdi","year":"2019"},{"key":"e_1_3_2_2_74_1","doi-asserted-by":"crossref","first-page":"2865","DOI":"10.1109\/TPDS.2020.3004623","article-title":"GPGPU Performance Estimation With Core and Memory Frequency Scaling","volume":"31","author":"Wang Qiang","year":"2020","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"e_1_3_2_2_75_1","volume-title":"Proceedings of the 7th Symposium on Operating Systems Design and Implementation (OSDI '06)","author":"Weil Sage A.","year":"2006"},{"key":"e_1_3_2_2_76_1","first-page":"1","article-title":"What ' s working in HPC : Investigating HPC User Behavior and Productivity","volume":"2","author":"Wolter Nicole","year":"2006","journal-title":"CTWatch Quarterly"},{"key":"e_1_3_2_2_77_1","volume-title":"Gandiva: Introspective Cluster Scheduling for Deep Learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI '18)","author":"Xiao Wencong","year":"2018"},{"key":"e_1_3_2_2_78_1","volume-title":"AntMan: Dynamic Scaling on GPU Clusters for Deep Learning. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI '20)","author":"Xiao Wencong","year":"2020"},{"key":"e_1_3_2_2_79_1","volume-title":"SLURM: Simple Linux Utility for Resource Management. In Job Scheduling Strategies for Parallel Processing.","author":"Yoo Andy B.","year":"2003"},{"key":"e_1_3_2_2_80_1","first-page":"1","article-title":"Burstiness-aware service level planning for enterprise application clouds","volume":"6","author":"Youssef Anas A.","year":"2017","journal-title":"Journal of Cloud Computing"},{"key":"e_1_3_2_2_81_1","volume-title":"Proceedings of Machine Learning and Systems (MLSys '20)","author":"Yu Peifeng","year":"2020"},{"key":"e_1_3_2_2_82_1","volume-title":"Proceedings of the ACM\/IEEE 42nd International Conference on Software Engineering (ICSE '20)","author":"Zhang Ru","year":"2020"},{"key":"e_1_3_2_2_83_1","doi-asserted-by":"crossref","first-page":"964","DOI":"10.1109\/TPDS.2015.2425403","article-title":"Burstiness-Aware Resource Reservation for Server Consolidation in Computing Clouds","volume":"27","author":"Zhang Sheng","year":"2016","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"e_1_3_2_2_84_1","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation (OSDI '20)","author":"Zhao Hanyu","year":"2020"}],"event":{"name":"SC '21: The International Conference for High Performance Computing, Networking, Storage and Analysis","location":"St. Louis Missouri","acronym":"SC '21","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","IEEE CS"]},"container-title":["Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3458817.3476223","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3458817.3476223","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:12:22Z","timestamp":1750191142000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3458817.3476223"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,11,13]]},"references-count":84,"alternative-id":["10.1145\/3458817.3476223","10.1145\/3458817"],"URL":"https:\/\/doi.org\/10.1145\/3458817.3476223","relation":{},"subject":[],"published":{"date-parts":[[2021,11,13]]}}}