{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T23:35:55Z","timestamp":1783035355928,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":85,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,3,30]],"date-time":"2025-03-30T00:00:00Z","timestamp":1743292800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nd\/4.0\/"}],"funder":[{"name":"Samsung Research Funding & Incubation Center of Samsung Electronics","award":["SRFC-IT1901-53"],"award-info":[{"award-number":["SRFC-IT1901-53"]}]},{"name":"MSIT (Ministry of Science and ICT), Korea","award":["IITP-2024-2021-0-01817"],"award-info":[{"award-number":["IITP-2024-2021-0-01817"]}]},{"name":"Samsung SDS Co., Ltd"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,3,30]]},"DOI":"10.1145\/3689031.3696078","type":"proceedings-article","created":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T06:25:20Z","timestamp":1742970320000},"page":"769-786","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["JABAS: Joint Adaptive Batching and Automatic Scaling for DNN Training on Heterogeneous GPUs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-8725-1534","authenticated-orcid":false,"given":"Gyeongchan","family":"Yun","sequence":"first","affiliation":[{"name":"UNIST"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6020-6744","authenticated-orcid":false,"given":"Junesoo","family":"Kang","sequence":"additional","affiliation":[{"name":"UNIST"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-2628-9811","authenticated-orcid":false,"given":"Hyunjoon","family":"Jeong","sequence":"additional","affiliation":[{"name":"UNIST"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-9545-9619","authenticated-orcid":false,"given":"Sanghyeon","family":"Eom","sequence":"additional","affiliation":[{"name":"UNIST"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-7805-6383","authenticated-orcid":false,"given":"Minsung","family":"Jang","sequence":"additional","affiliation":[{"name":"Samsung SDS"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4391-4470","authenticated-orcid":false,"given":"Young-ri","family":"Choi","sequence":"additional","affiliation":[{"name":"UNIST"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,3,30]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2021. https:\/\/github.com\/SimiGrad\/SimiGrad."},{"key":"e_1_3_2_1_2_1","volume-title":"https:\/\/github.com\/sUntvoOk\/EasyScale_info_for_SC23","unstructured":"2023. https:\/\/github.com\/sUntvoOk\/EasyScale_info_for_SC23."},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of Machine Learning and Systems (MLSys).","author":"Agarwal Saurabh","year":"2021","unstructured":"Saurabh Agarwal, Hongyi Wang, Kangwook Lee, Shivaram Venkataraman, and Dimitris Papailiopoulos. 2021. Adaptive gradient communication via critical learning regime identification. In Proceedings of Machine Learning and Systems (MLSys)."},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the 14th USENIX Symposium on Networked Systems Design and Implementation (NSDI).","author":"Alipourfard Omid","year":"2017","unstructured":"Omid Alipourfard, Hongqiang Harry Liu, Jianshu Chen, Shivaram Venkataraman, Minlan Yu, and Ming Zhang. 2017. CherryPick: Adaptively unearthing the best cloud configurations for big data analytics. In Proceedings of the 14th USENIX Symposium on Networked Systems Design and Implementation (NSDI)."},{"key":"e_1_3_2_1_5_1","unstructured":"AWS. 2023. AWS P3 Instance. https:\/\/aws.amazon.com\/ec2\/instance-types\/p3."},{"key":"e_1_3_2_1_6_1","volume-title":"Findings of the 2016 conference on machine translation (WMT16). In Proceedings of the First Conference on Machine Translation.","author":"Bojar Ondrej","year":"2016","unstructured":"Ondrej Bojar, Rajen Chatterjee, Christian Federmann, Yvette Graham, Barry Haddow, Matthias Huck, Antonio Jimeno Yepes, Philipp Koehn, Varvara Logacheva, Christof Monz, et al. 2016. Findings of the 2016 conference on machine translation (WMT16). In Proceedings of the First Conference on Machine Translation."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1080\/01621459.1970.10481180"},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems (NeurIPS).","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, et al. 2020. Language models are few-shot learners. In Proceedings of the Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3342195.3387555"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421299"},{"key":"e_1_3_2_1_11_1","unstructured":"Sharan Chetlur Cliff Woolley Philippe Vandermersch Jonathan Cohen John Tran Bryan Catanzaro and Evan Shelhamer. 2014. cuDNN: Efficient primitives for deep learning. arXiv preprint arXiv:1410.0759."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.2517-6161.1958.tb00292.x"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_14_1","volume-title":"BERT: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805.","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. BERT: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421284"},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings of the 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI).","author":"Duan Jiangfei","year":"2024","unstructured":"Jiangfei Duan, Ziang Song, Xupeng Miao, Xiaoli Xi, Dahua Lin, Harry Xu, Minjia Zhang, and Zhihao Jia. 2024. Parcae: Proactive, Liveput-Optimized DNN Training on Preemptible Instances. In Proceedings of the 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI)."},{"key":"e_1_3_2_1_17_1","unstructured":"Fartash Faghri David Duvenaud David J Fleet and Jimmy Ba. 2020. A Study of Gradient Variance in Deep Learning. arXiv preprint arXiv:2007.04532."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1002\/for.3980040103"},{"key":"e_1_3_2_1_19_1","unstructured":"Priya Goyal Piotr Doll\u00e1r Ross Girshick Pieter Noordhuis Lukasz Wesolowski Aapo Kyrola Andrew Tulloch Yangqing Jia and Kaiming He. 2017. Accurate large minibatch sgd: Training imagenet in 1 hour. arXiv preprint arXiv:1706.02677."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575721"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long short-term memory. Neural computation 9.","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_2_1_23_1","volume-title":"proceedings of Advances in neural information processing systems (NeurIPS).","author":"Hoffer Elad","year":"2017","unstructured":"Elad Hoffer, Itay Hubara, and Daniel Soudry. 2017. Train longer, generalize better: closing the generalization gap in large batch training of neural networks. In proceedings of Advances in neural information processing systems (NeurIPS)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575705"},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of the 18th USENIX Symposium on Networked Systems Design and Implementation (NSDI).","author":"Hwang Changho","year":"2021","unstructured":"Changho Hwang, Taehyun Kim, Sunghyun Kim, Jinwoo Shin, and KyoungSoo Park. 2021. Elastic resource sharing for distributed deep learning. In Proceedings of the 18th USENIX Symposium on Networked Systems Design and Implementation (NSDI)."},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of the 32nd International Conference on Machine Learning (ICML).","author":"Ioffe Sergey","year":"2015","unstructured":"Sergey Ioffe and Christian Szegedy. 2015. Batch Normalization: Accelerating deep network training by reducing internal covariate shift. In Proceedings of the 32nd International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613152"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613175"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.5555\/3358807.3358888"},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the 2022 USENIX Annual Technical Conference (ATC).","author":"Jia Xianyan","year":"2022","unstructured":"Xianyan Jia, Le Jiang, Ang Wang, Wencong Xiao, Ziji Shi, Jie Zhang, Xinyuan Li, Langshi Chen, Yong Li, Zhen Zheng, et al. 2022. Whale: Efficient Giant Model Training over Heterogeneous GPUs. In Proceedings of the 2022 USENIX Annual Technical Conference (ATC)."},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the USENIX Symposium on Operating Systems Design and Implementation (OSDI).","author":"Jiang Yimin","year":"2020","unstructured":"Yimin Jiang, Yibo Zhu, Chang Lan, Bairen Yi, Yong Cui, and Chuanxiong Guo. 2020. A unified architecture for accelerating distributed DNN training in heterogeneous GPU\/CPU clusters. In Proceedings of the USENIX Symposium on Operating Systems Design and Implementation (OSDI)."},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the 37th International Conference on Machine Learning (ICML).","author":"Johnson Tyler B","year":"2020","unstructured":"Tyler B Johnson, Pulkit Agrawal, Haijie Gu, and Carlos Guestrin. 2020. AdaScale SGD: A Scale-Invariant Algorithm for Distributed Training. In Proceedings of the 37th International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_33_1","volume-title":"Scale-Train: A Scalable DNN Training Framework for a Heterogeneous GPU Cloud","author":"Kim Kyeonglok","unstructured":"Kyeonglok Kim, Hyeonsu Lee, Seungmin Oh, and Euiseong Seo. 2022. Scale-Train: A Scalable DNN Training Framework for a Heterogeneous GPU Cloud. IEEE Access."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3302424.3303957"},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML).","author":"Kleinberg Bobby","year":"2018","unstructured":"Bobby Kleinberg, Yuanzhi Li, and Yang Yuan. 2018. An alternative view: When does SGD escape local minima?. In Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.14778\/3342263.3342276"},{"key":"e_1_3_2_1_37_1","volume-title":"One weird trick for parallelizing convolutional neural networks. arXiv preprint arXiv:1404.5997","author":"Krizhevsky Alex","year":"2014","unstructured":"Alex Krizhevsky. 2014. One weird trick for parallelizing convolutional neural networks. arXiv preprint arXiv:1404.5997 (2014)."},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems (NeurIPS).","author":"Krizhevsky Alex","year":"2012","unstructured":"Alex Krizhevsky, Ilya Sutskever, and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks. In Proceedings of the Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_39_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems (NeurIPS).","author":"Kwon Woosuk","year":"2020","unstructured":"Woosuk Kwon, Gyeong-In Yu, Eunji Jeong, and Byung-Gon Chun. 2020. Nimble: Lightweight and parallel gpu task scheduling for deep learning. In Proceedings of the Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_40_1","unstructured":"Simon Lacoste-Julien Mark Schmidt and Francis Bach. 2012. A simpler approach to obtaining an O (1\/t) convergence rate for the projected stochastic subgradient method. arXiv preprint arXiv:1212.2002."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3552326.3587445"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607054"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.14778\/3415478.3415530"},{"key":"e_1_3_2_1_44_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML).","author":"Lian Xiangru","year":"2018","unstructured":"Xiangru Lian, Wei Zhang, Ce Zhang, and Ji Liu. 2018. Asynchronous decentralized parallel stochastic gradient descent. In Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_45_1","volume-title":"Proceedings of the 2021 USENIX Annual Technical Conference (ATC).","author":"Lim Gangmuk","year":"2021","unstructured":"Gangmuk Lim, Jeongseob Ahn, Wencong Xiao, Youngjin Kwon, and Myeongjae Jeon. 2021. Zico: Efficient GPU Memory Sharing for Concurrent DNN Training. In Proceedings of the 2021 USENIX Annual Technical Conference (ATC)."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378499"},{"key":"e_1_3_2_1_48_1","volume-title":"Proceedings of the 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI).","author":"Mai Luo","year":"2020","unstructured":"Luo Mai, Guo Li, Marcel Wagenl\u00e4nder, Konstantinos Fertakis, Andrei-Octavian Brabete, and Peter Pietzuch. 2020. KungFu: Making training in distributed machine learning adaptive. In Proceedings of the 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI)."},{"key":"e_1_3_2_1_49_1","unstructured":"Sam McCandlish Jared Kaplan Dario Amodei and OpenAI Dota Team. 2018. An empirical model of large-batch training. arXiv preprint arXiv:1812.06162."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.14778\/3598581.3598604"},{"key":"e_1_3_2_1_51_1","volume-title":"Proceeding of the USENIX Symposium on Operating Systems Design and Implementation (OSDI).","author":"Narayanan Deepak","year":"2020","unstructured":"Deepak Narayanan, Keshav Santhanam, Fiodar Kazhamiaka, Amar Phanishayee, and Matei Zaharia. 2020. Heterogeneity-Aware Cluster Scheduling Policies for Deep Learning Workloads. In Proceeding of the USENIX Symposium on Operating Systems Design and Implementation (OSDI)."},{"key":"e_1_3_2_1_52_1","volume-title":"Jianyu Huang, Narayanan Sundaraman, Jongsoo Park, Xiaodong Wang, Udit Gupta, Carole-Jean Wu, Alisson G Azzolini, et al.","author":"Naumov Maxim","year":"2019","unstructured":"Maxim Naumov, Dheevatsa Mudigere, Hao-Jun Michael Shi, Jianyu Huang, Narayanan Sundaraman, Jongsoo Park, Xiaodong Wang, Udit Gupta, Carole-Jean Wu, Alisson G Azzolini, et al. 2019. Deep learning recommendation model for personalization and recommendation systems. arXiv preprint arXiv:1906.00091."},{"key":"e_1_3_2_1_53_1","unstructured":"NVIDIA corp. NVIDIA Multi-Instance GPU (MIG). https:\/\/docs.nvidia.com\/cuda\/pdf\/MIG_User_Guide.pdf."},{"key":"e_1_3_2_1_54_1","unstructured":"NVIDIA corp. NVIDIA Multi-Processes Service (MPS). https:\/\/docs.nvidia.com\/deploy\/pdf\/CUDA_Multi_Process_Service_Overview.pdf."},{"key":"e_1_3_2_1_55_1","unstructured":"NVIDIA corp.. NVIDIA NCCL. https:\/\/developer.nvidia.com\/nccl."},{"key":"e_1_3_2_1_56_1","unstructured":"NVIDIA corp. NVIDIA NCCL error. https:\/\/docs.nvidia.com\/deeplearning\/nccl\/user-guide\/docs\/usage\/communicators.html#error-handling-and-communicator-abort."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3492321.3519563"},{"key":"e_1_3_2_1_58_1","volume-title":"Proceedings of Machine Learning and Systems (MLSys).","author":"Or Andrew","year":"2020","unstructured":"Andrew Or, Haoyu Zhang, and Michael Freedman. 2020. Resource elasticity in distributed deep learning. In Proceedings of Machine Learning and Systems (MLSys)."},{"key":"e_1_3_2_1_59_1","volume-title":"Proceedings of Machine Learning and Systems (MLSys).","author":"Or Andrew","year":"2022","unstructured":"Andrew Or, Haoyu Zhang, and Michael None Freedman. 2022. VirtualFlow: Decoupling Deep Learning Models from the Underlying Hardware. In Proceedings of Machine Learning and Systems (MLSys)."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359642"},{"key":"e_1_3_2_1_61_1","unstructured":"Xin Qian and Diego Klabjan. 2020. The impact of the mini-batch size on the variance of gradients in stochastic gradient descent. arXiv preprint arXiv:2004.13146."},{"key":"e_1_3_2_1_62_1","volume-title":"Proceedings of the 15th USENIX Symposium on Operating Systems Design and Implementation (OSDI).","author":"Qiao Aurick","unstructured":"Aurick Qiao, Sang Keun Choe, Suhas Jayaram Subramanya, Willie Neiswanger, Qirong Ho, Hao Zhang, Gregory R. Ganger, and Eric P. Xing. 2021. Pollux: Co-adaptive Cluster Scheduling for Goodput-Optimized Deep Learning. In Proceedings of the 15th USENIX Symposium on Operating Systems Design and Implementation (OSDI)."},{"key":"e_1_3_2_1_63_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems (NeurIPS).","author":"Qin Heyang","year":"2021","unstructured":"Heyang Qin, Samyam Rajbhandari, Olatunji Ruwase, Feng Yan, Lei Yang, and Yuxiong He. 2021. SimiGrad: Fine-grained adaptive batching for large scale training using gradient similarity measurement. In Proceedings of the Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_64_1","unstructured":"Alec Radford Jeff Wu Rewon Child David Luan Dario Amodei and Ilya Sutskever. 2019. Language Models are Unsupervised Multitask Learners."},{"key":"e_1_3_2_1_65_1","volume-title":"YOLOv3: An Incremental Improvement. arXiv preprint arXiv:1804.02767","author":"Redmon Joseph","year":"2018","unstructured":"Joseph Redmon and Ali Farhadi. 2018. YOLOv3: An Incremental Improvement. arXiv preprint arXiv:1804.02767 (2018)."},{"key":"e_1_3_2_1_66_1","volume-title":"Proceedings of the 2021 USENIX Annual Technical Conference (ATC).","author":"Ren Jie","year":"2021","unstructured":"Jie Ren, Samyam Rajbhandari, Reza Yazdani Aminabadi, Olatunji Ruwase, Shuangyan Yang, Minjia Zhang, Dong Li, and Yuxiong He. 2021. ZeRO-Offload: Democratizing Billion-Scale Model Training. In Proceedings of the 2021 USENIX Annual Technical Conference (ATC)."},{"key":"e_1_3_2_1_67_1","volume-title":"proceedings of Advances in Neural Information Processing Systems (NeurIPS).","author":"Ren Shaoqing","year":"2015","unstructured":"Shaoqing Ren, Kaiming He, Ross Girshick, and Jian Sun. 2015. Faster R-CNN: Towards real-time object detection with region proposal networks. In proceedings of Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_68_1","volume-title":"Proceedings of the USENIX Symposium on Networked Systems Design and Implementation (NSDI).","author":"Romero Joshua","year":"2022","unstructured":"Joshua Romero, Junqi Yin, Nouamane Laanait, Bing Xie, M Todd Young, Sean Treichler, Vitalii Starchenko, Albina Borisevich, Alex Sergeev, and Michael Matheson. 2022. Accelerating Collective Communication in Data Parallel Training across Deep Learning Frameworks. In Proceedings of the USENIX Symposium on Networked Systems Design and Implementation (NSDI)."},{"key":"e_1_3_2_1_69_1","unstructured":"Alexander Sergeev and Mike Del Balso. 2018. Horovod: fast and easy distributed deep learning in TensorFlow. arXiv preprint arXiv:1802.05799."},{"key":"e_1_3_2_1_70_1","volume-title":"Proceedings of the 2020 USENIX Annual Technical Conference (ATC).","author":"Shahrad Mohammad","year":"2020","unstructured":"Mohammad Shahrad, Rodrigo Fonseca, Inigo Goiri, Gohar Chaudhry, Paul Batum, Jason Cooke, Eduardo Laureano, Colby Tresness, Mark Russinovich, and Ricardo Bianchini. 2020. Serverless in the wild: Characterizing and optimizing the serverless workload at a large cloud provider. In Proceedings of the 2020 USENIX Annual Technical Conference (ATC)."},{"key":"e_1_3_2_1_71_1","article-title":"Measuring the effects of data parallelism on neural network training","author":"Shallue Christopher J","year":"2019","unstructured":"Christopher J Shallue, Jaehoon Lee, Joseph Antognini, Jascha Sohl-Dickstein, Roy Frostig, and George E Dahl. 2019. Measuring the effects of data parallelism on neural network training. Journal of Machine Learning Research 20.","journal-title":"Journal of Machine Learning Research 20."},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177728435"},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.1145\/3297858.3304072"},{"key":"e_1_3_2_1_74_1","unstructured":"Samuel L Smith Pieter-Jan Kindermans Chris Ying and Quoc V Le. 2017. Don't decay the learning rate increase the batch size. arXiv preprint arXiv:1711.00489."},{"key":"e_1_3_2_1_75_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288."},{"key":"e_1_3_2_1_76_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems (NeurIPS).","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention Is All You Need. In Proceedings of the Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_77_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems (NeurIPS).","author":"Wang Chong","year":"2013","unstructured":"Chong Wang, Xi Chen, Alexander J Smola, and Eric P Xing. 2013. Variance reduction for stochastic gradient optimization. In Proceedings of the Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_78_1","volume-title":"Proceedings of the USENIX Symposium on Networked Systems Design and Implementation (NSDI).","author":"Weng Qizhen","year":"2022","unstructured":"Qizhen Weng, Wencong Xiao, Yinghao Yu, Wei Wang, Cheng Wang, Jian He, Yong Li, Liping Zhang, Wei Lin, and Yu Ding. 2022. MLaaS in the Wild: Workload Analysis and Scheduling in Large-Scale Heterogeneous GPU Clusters. In Proceedings of the USENIX Symposium on Networked Systems Design and Implementation (NSDI)."},{"key":"e_1_3_2_1_79_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems (NeurIPS).","author":"Williams Christopher","year":"1995","unstructured":"Christopher Williams and Carl Rasmussen. 1995. Gaussian processes for regression. In Proceedings of the Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_80_1","volume-title":"Roofline: an insightful visual performance model for multicore architectures. Commun. ACM 52","author":"Williams Samuel","unstructured":"Samuel Williams, Andrew Waterman, and David Patterson. 2009. Roofline: an insightful visual performance model for multicore architectures. Commun. ACM 52."},{"key":"e_1_3_2_1_81_1","unstructured":"Yonghui Wu Mike Schuster Zhifeng Chen Quoc V Le Mohammad Norouzi Wolfgang Macherey Maxim Krikun Yuan Cao Qin Gao Klaus Macherey et al. 2016. Google's neural machine translation system: Bridging the gap between human and machine translation. arXiv preprint arXiv:1609.08144."},{"key":"e_1_3_2_1_82_1","volume-title":"DBS: Dynamic batch size for distributed deep neural network training. arXiv preprint arXiv:2007.11831.","author":"Ye Qing","year":"2020","unstructured":"Qing Ye, Yuhao Zhou, Mingjia Shi, Yanan Sun, and Jiancheng Lv. 2020. DBS: Dynamic batch size for distributed deep neural network training. arXiv preprint arXiv:2007.11831."},{"key":"e_1_3_2_1_83_1","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV).","author":"Yuan Li","year":"2021","unstructured":"Li Yuan, Yunpeng Chen, Tao Wang, Weihao Yu, Yujun Shi, Zi-Hang Jiang, Francis EH Tay, Jiashi Feng, and Shuicheng Yan. 2021. Tokensto-Token ViT: Training vision transformers from scratch on imagenet. In Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)."},{"key":"e_1_3_2_1_84_1","volume-title":"Proceedings of the 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI).","author":"Zheng Pengfei","year":"2023","unstructured":"Pengfei Zheng, Rui Pan, Tarannum Khan, Shivaram Venkataraman, and Aditya Akella. 2023. Shockwave: Fair and Efficient Cluster Scheduling for Dynamic Adaptation in Machine Learning. In Proceedings of the 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI)."},{"key":"e_1_3_2_1_85_1","volume-title":"Proceedings of the 2022 USENIX Annual Technical Conference (ATC).","author":"Zhou Zhe","year":"2022","unstructured":"Zhe Zhou, Xuechao Wei, Jiejing Zhang, and Guangyu Sun. 2022. PetS: A Unified Framework for Parameter-Efficient Transformers Serving. In Proceedings of the 2022 USENIX Annual Technical Conference (ATC)."}],"event":{"name":"EuroSys '25: Twentieth European Conference on Computer Systems","location":"Rotterdam Netherlands","acronym":"EuroSys '25","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the Twentieth European Conference on Computer Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3689031.3696078","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3689031.3696078","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T11:19:26Z","timestamp":1755775166000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3689031.3696078"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,30]]},"references-count":85,"alternative-id":["10.1145\/3689031.3696078","10.1145\/3689031"],"URL":"https:\/\/doi.org\/10.1145\/3689031.3696078","relation":{},"subject":[],"published":{"date-parts":[[2025,3,30]]},"assertion":[{"value":"2025-03-30","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}