{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:10:14Z","timestamp":1777655414913,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,4,25]],"date-time":"2022-04-25T00:00:00Z","timestamp":1650844800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["2020R1A2B5B03001960"],"award-info":[{"award-number":["2020R1A2B5B03001960"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Institute of Information & Communications Technology Planning & Evaluation","award":["2020-0-01373"],"award-info":[{"award-number":["2020-0-01373"]}]},{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["1820609"],"award-info":[{"award-number":["1820609"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,4,25]]},"DOI":"10.1145\/3485447.3511989","type":"proceedings-article","created":{"date-parts":[[2022,4,25]],"date-time":"2022-04-25T05:13:07Z","timestamp":1650863587000},"page":"1851-1859","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["Not All Layers Are Equal: A Layer-Wise Adaptive Approach Toward Large-Scale DNN Training"],"prefix":"10.1145","author":[{"given":"Yunyong","family":"Ko","sequence":"first","affiliation":[{"name":"Hanyang University, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dongwon","family":"Lee","sequence":"additional","affiliation":[{"name":"The Pennsylvania State University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sang-Wook","family":"Kim","sequence":"additional","affiliation":[{"name":"Hanyang University, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,4,25]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML). PMLR, 344\u2013353","author":"Assran Mahmoud","year":"2019","unstructured":"Mahmoud Assran, Nicolas Loizou, Nicolas Ballas, and Mike Rabbat. 2019. Stochastic gradient push for distributed deep learning. In Proceedings of the International Conference on Machine Learning (ICML). PMLR, 344\u2013353."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Medha Atre Birendra Jha and Ashwini Rao. 2021. Distributed Deep Learning Using Volunteer Computing-Like Paradigm. arXiv preprint arXiv:2103.08894(2021).","DOI":"10.1109\/IPDPSW52791.2021.00144"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1907375117"},{"key":"e_1_3_2_1_4_1","unstructured":"Oded Ben-David and Zohar Ringel. 2019. The role of a layer in deep neural networks: a Gaussian Process perspective. arXiv preprint arXiv:1902.02354(2019)."},{"key":"e_1_3_2_1_5_1","unstructured":"Tom\u00a0B Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell 2020. Language models are few-shot learners. arXiv preprint arXiv:2005.14165(2020)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3308558.3313413"},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the Symposium on Operating Systems Design and Implementation (OSDI)). 571\u2013582","author":"Chilimbi Trishul","year":"2014","unstructured":"Trishul Chilimbi, Yutaka Suzue, Johnson Apacible, and Karthik Kalyanaraman. 2014. Project adam: Building an efficient and scalable deep learning training system. In Proceedings of the Symposium on Operating Systems Design and Implementation (OSDI)). 571\u2013582."},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems. 1223\u20131231","author":"Dean Jeffrey","year":"2012","unstructured":"Jeffrey Dean, Greg Corrado, Rajat Monga, Kai Chen, Matthieu Devin, Mark Mao, Andrew Senior, Paul Tucker, Ke Yang, Quoc\u00a0V Le, 2012. Large scale distributed deep networks. In Proceedings of the Advances in Neural Information Processing Systems. 1223\u20131231."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_10_1","volume-title":"Adabatch: Adaptive batch sizes for training deep neural networks. arXiv preprint arXiv:1712.02029(2017).","author":"Devarakonda Aditya","year":"2017","unstructured":"Aditya Devarakonda, Maxim Naumov, and Michael Garland. 2017. Adabatch: Adaptive batch sizes for training deep neural networks. arXiv preprint arXiv:1712.02029(2017)."},{"key":"e_1_3_2_1_11_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805(2018).","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805(2018)."},{"key":"e_1_3_2_1_12_1","unstructured":"Priya Goyal Piotr Doll\u00e1r Ross Girshick Pieter Noordhuis Lukasz Wesolowski Aapo Kyrola Andrew Tulloch Yangqing Jia and Kaiming He. 2017. Accurate large minibatch sgd: Training imagenet in 1 hour. arXiv preprint arXiv:1706.02677(2017)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems. 1223\u20131231","author":"Ho Qirong","year":"2013","unstructured":"Qirong Ho, James Cipar, Henggang Cui, Seunghak Lee, Jin\u00a0Kyu Kim, Phillip\u00a0B Gibbons, Garth\u00a0A Gibson, Greg Ganger, and Eric\u00a0P Xing. 2013. More effective distributed ml via a stale synchronous parallel parameter server. In Proceedings of the Advances in Neural Information Processing Systems. 1223\u20131231."},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the International Conference on Neural Information Processing Systems (NeurIPS). 1729\u20131739","author":"Hoffer Elad","year":"2017","unstructured":"Elad Hoffer, Itay Hubara, and Daniel Soudry. 2017. Train longer, generalize better: closing the generalization gap in large batch training of neural networks. In Proceedings of the International Conference on Neural Information Processing Systems (NeurIPS). 1729\u20131739."},{"key":"e_1_3_2_1_16_1","unstructured":"Zhouyuan Huo Bin Gu and Heng Huang. 2020. Large batch training does not need warmup. arXiv preprint arXiv:2002.01576(2020)."},{"key":"e_1_3_2_1_17_1","volume-title":"Parallelizing stochastic gradient descent for least squares regression: mini-batching, averaging, and model misspecification. Journal of Machine Learning Research 18","author":"Jain Prateek","year":"2018","unstructured":"Prateek Jain, Sham Kakade, Rahul Kidambi, Praneeth Netrapalli, and Aaron Sidford. 2018. Parallelizing stochastic gradient descent for least squares regression: mini-batching, averaging, and model misspecification. Journal of Machine Learning Research 18 (2018)."},{"key":"e_1_3_2_1_18_1","unstructured":"Stanis\u0142aw Jastrz\u0119bski Zachary Kenton Devansh Arpit Nicolas Ballas Asja Fischer Yoshua Bengio and Amos Storkey. 2017. Three factors influencing minima in sgd. arXiv preprint arXiv:1711.04623(2017)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3308558.3313564"},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML). PMLR, 4911\u20134920","author":"Johnson Tyler","year":"2020","unstructured":"Tyler Johnson, Pulkit Agrawal, Haijie Gu, and Carlos Guestrin. 2020. AdaScale SGD: A User-Friendly Algorithm for Distributed Training. In Proceedings of the International Conference on Machine Learning (ICML). PMLR, 4911\u20134920."},{"key":"e_1_3_2_1_21_1","first-page":"7265","article-title":"Training deep models faster with robust, approximate importance sampling","volume":"31","author":"Johnson B","year":"2018","unstructured":"Tyler\u00a0B Johnson and Carlos Guestrin. 2018. Training deep models faster with robust, approximate importance sampling. Advances in Neural Information Processing Systems 31 (2018), 7265\u20137275.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_22_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980(2014).","author":"Kingma P","year":"2014","unstructured":"Diederik\u00a0P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980(2014)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3482412"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS49936.2021.00108"},{"key":"e_1_3_2_1_25_1","unstructured":"Alex Krizhevsky. 2014. One weird trick for parallelizing convolutional neural networks. arXiv preprint arXiv:1404.5997(2014)."},{"key":"e_1_3_2_1_26_1","unstructured":"Alex Krizhevsky Geoffrey Hinton 2009. Learning multiple layers of features from tiny images. (2009)."},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the World Wide Web Conference (WWW). 783\u2013792","author":"Lee Youngnam","year":"2018","unstructured":"Youngnam Lee, Sang-Wook Kim, Sunju Park, and Xing Xie. 2018. How to impute missing ratings? Claims, solution, and its application to collaborative filtering. In Proceedings of the World Wide Web Conference (WWW). 783\u2013792."},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML). PMLR, 3325\u20133334","author":"Ma Siyuan","year":"2018","unstructured":"Siyuan Ma, Raef Bassily, and Mikhail Belkin. 2018. The power of interpolation: Understanding the effectiveness of SGD in modern over-parametrized learning. In Proceedings of the International Conference on Machine Learning (ICML). PMLR, 3325\u20133334."},{"key":"e_1_3_2_1_29_1","unstructured":"Sam McCandlish Jared Kaplan Dario Amodei and OpenAI\u00a0Dota Team. 2018. An empirical model of large-batch training. arXiv preprint arXiv:1812.06162(2018)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442381.3449815"},{"key":"e_1_3_2_1_31_1","volume-title":"Reliability of Large Scale GPU Clusters for Deep Learning Workloads. In Companion Proceedings of the Web Conference (WWW). 179\u2013181","author":"Qian Junjie","year":"2021","unstructured":"Junjie Qian, Taeyoon Kim, and Myeongjae Jeon. 2021. Reliability of Large Scale GPU Clusters for Deep Learning Workloads. In Companion Proceedings of the Web Conference (WWW). 179\u2013181."},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems (NIPS). 693\u2013701","author":"Recht Benjamin","year":"2011","unstructured":"Benjamin Recht, Christopher Re, Stephen Wright, and Feng Niu. 2011. Hogwild: A lock-free approach to parallelizing stochastic gradient descent. In Proceedings of the Advances in Neural Information Processing Systems (NIPS). 693\u2013701."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_34_1","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR).","author":"Simonyan Karen","year":"2015","unstructured":"Karen Simonyan and Andrew Zisserman. 2015. Very deep convolutional networks for large-scale image recognition. In Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of International Conference on Learning Representations (ICLR).","author":"Smith L","year":"2018","unstructured":"Samuel\u00a0L Smith, Pieter-Jan Kindermans, Chris Ying, and Quoc\u00a0V Le. 2018. Don\u2019t Decay the Learning Rate, Increase the Batch Size. In Proceedings of International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_36_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML). PMLR, 6105\u20136114","author":"Tan Mingxing","year":"2019","unstructured":"Mingxing Tan and Quoc Le. 2019. Efficientnet: Rethinking model scaling for convolutional neural networks. In Proceedings of the International Conference on Machine Learning (ICML). PMLR, 6105\u20136114."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2020.2974267"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/2783258.2783323"},{"key":"e_1_3_2_1_39_1","volume-title":"Parallelizing DNN Training on GPUs: Challenges and Opportunities. In Companion Proceedings of the Web Conference (WWW). 174\u2013178","author":"Xu Weizheng","year":"2021","unstructured":"Weizheng Xu, Youtao Zhang, and Xulong Tang. 2021. Parallelizing DNN Training on GPUs: Challenges and Opportunities. In Companion Proceedings of the Web Conference (WWW). 174\u2013178."},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of the International Conference on Artificial Intelligence and Statistics (AISTATS). PMLR","author":"Yin Dong","year":"2018","unstructured":"Dong Yin, Ashwin Pananjady, Max Lam, Dimitris Papailiopoulos, Kannan Ramchandran, and Peter Bartlett. 2018. Gradient diversity: a key ingredient for scalable distributed learning. In Proceedings of the International Conference on Artificial Intelligence and Statistics (AISTATS). PMLR, 1998\u20132007."},{"key":"e_1_3_2_1_41_1","unstructured":"Yang You Igor Gitman and Boris Ginsburg. 2017. Large batch training of convolutional networks. arXiv preprint arXiv:1708.03888(2017)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356137"},{"key":"e_1_3_2_1_43_1","unstructured":"Yang You Jing Li Sashank Reddi Jonathan Hseu Sanjiv Kumar Srinadh Bhojanapalli Xiaodan Song James Demmel Kurt Keutzer and Cho-Jui Hsieh. 2019. Large batch optimization for deep learning: Training bert in 76 minutes. arXiv preprint arXiv:1904.00962(2019)."},{"key":"e_1_3_2_1_44_1","volume-title":"Proceedings of the USENIX Annual Technical Conference (ATC). 181\u2013193","author":"Zhang Hao","year":"2017","unstructured":"Hao Zhang, Zeyu Zheng, Shizhen Xu, Wei Dai, Qirong Ho, Xiaodan Liang, Zhiting Hu, Jinliang Wei, Pengtao Xie, and Eric\u00a0P Xing. 2017. Poseidon: An Efficient Communication Architecture for Distributed Deep Learning on GPU Clusters. In Proceedings of the USENIX Annual Technical Conference (ATC). 181\u2013193."},{"key":"e_1_3_2_1_45_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems. 685\u2013693","author":"Zhang Sixin","year":"2015","unstructured":"Sixin Zhang, Anna\u00a0E Choromanska, and Yann LeCun. 2015. Deep learning with elastic averaging SGD. In Proceedings of the Advances in Neural Information Processing Systems. 685\u2013693."},{"key":"e_1_3_2_1_46_1","volume-title":"International Conference on Machine Learning (ICML). PMLR, 4120\u20134129","author":"Zheng Shuxin","year":"2017","unstructured":"Shuxin Zheng, Qi Meng, Taifeng Wang, Wei Chen, Nenghai Yu, Zhi-Ming Ma, and Tie-Yan Liu. 2017. Asynchronous stochastic gradient descent with delay compensation. In International Conference on Machine Learning (ICML). PMLR, 4120\u20134129."}],"event":{"name":"WWW '22: The ACM Web Conference 2022","location":"Virtual Event, Lyon France","acronym":"WWW '22","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM Web Conference 2022"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3485447.3511989","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3485447.3511989","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3485447.3511989","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:12:10Z","timestamp":1750191130000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3485447.3511989"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,4,25]]},"references-count":46,"alternative-id":["10.1145\/3485447.3511989","10.1145\/3485447"],"URL":"https:\/\/doi.org\/10.1145\/3485447.3511989","relation":{},"subject":[],"published":{"date-parts":[[2022,4,25]]},"assertion":[{"value":"2022-04-25","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}