{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T10:59:29Z","timestamp":1777546769297,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":35,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,2,21]],"date-time":"2023-02-21T00:00:00Z","timestamp":1676937600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62272168"],"award-info":[{"award-number":["62272168"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,2,25]]},"DOI":"10.1145\/3572848.3577484","type":"proceedings-article","created":{"date-parts":[[2023,2,21]],"date-time":"2023-02-21T16:02:30Z","timestamp":1676995350000},"page":"380-391","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":9,"title":["Elastic Averaging for Efficient Pipelined DNN Training"],"prefix":"10.1145","author":[{"given":"Zihao","family":"Chen","sequence":"first","affiliation":[{"name":"East China Normal University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chen","family":"Xu","sequence":"additional","affiliation":[{"name":"East China Normal University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weining","family":"Qian","sequence":"additional","affiliation":[{"name":"East China Normal University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aoying","family":"Zhou","sequence":"additional","affiliation":[{"name":"East China Normal University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,2,21]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"MXNet. https:\/\/mxnet.apache.org\/versions\/1.9.1."},{"key":"e_1_3_2_1_2_1","unstructured":"Penn Treebank. http:\/\/www.fit.vutbr.cz\/~imikolov\/rnnlm\/simple-examples.tgz."},{"key":"e_1_3_2_1_3_1","unstructured":"PyTorch. https:\/\/github.com\/pytorch\/pytorch."},{"key":"e_1_3_2_1_4_1","unstructured":"torchgpipe. https:\/\/torchgpipe.readthedocs.io\/en\/stable\/."},{"key":"e_1_3_2_1_5_1","unstructured":"WMT14 dataset. https:\/\/www.statmt.org\/wmt14\/translation-task.html."},{"key":"e_1_3_2_1_6_1","unstructured":"WMT16 dataset. https:\/\/www.statmt.org\/wmt16\/translation-task.html."},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the 12th USENIX Symposium on Operating Systems Design and Implementation (OSDI). 265--283","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi, Paul Barham, Jianmin Chen, Zhifeng Chen, Andy Davis, Jeffrey Dean, Matthieu Devin, Sanjay Ghemawat, Geoffrey Irving, Michael Isard, Manjunath Kudlur, Josh Levenberg, Rajat Monga, Sherry Moore, Derek Gordon Murray, Benoit Steiner, Paul A. Tucker, Vijay Vasudevan, Pete Warden, Martin Wicke, Yuan Yu, and Xiaoqiang Zheng. 2016. TensorFlow: A System for Large-Scale Machine Learning. In Proceedings of the 12th USENIX Symposium on Operating Systems Design and Implementation (OSDI). 265--283."},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems (NIPS)","volume":"33","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, Sandhini Agarwal, Ariel Herbert-Voss, Gretchen Krueger, Tom Henighan, Rewon Child, Aditya Ramesh, Daniel Ziegler, Jeffrey Wu, Clemens Winter, Chris Hesse, Mark Chen, Eric Sigler, Mateusz Litwin, Scott Gray, Benjamin Chess, Jack Clark, Christopher Berner, Sam McCandlish, Alec Radford, Ilya Sutskever, and Dario Amodei. 2020. Language Models are Few-Shot Learners. In Proceedings of the 34th International Conference on Neural Information Processing Systems (NIPS), Vol. 33. 1877--1901."},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL). 4171--4186","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL). 4171--4186."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.5555\/1953048.2021068"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441593"},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning (ICML)","volume":"139","author":"He Chaoyang","year":"2021","unstructured":"Chaoyang He, Shen Li, Mahdi Soltanolkotabi, and Salman Avestimehr. 2021. PipeTransformer: Automated Elastic Pipelining for Distributed Training of Large-scale Models. In Proceedings of the 38th International Conference on Machine Learning (ICML), Vol. 139. 4150--4159."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503221.3508418"},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the 33rd International Conference on Neural Information Processing Systems (NIPS)","volume":"32","author":"Huang Yanping","year":"2019","unstructured":"Yanping Huang, Youlong Cheng, Ankur Bapna, Orhan Firat, Dehao Chen, Mia Xu Chen, HyoukJoong Lee, Jiquan Ngiam, Quoc V. Le, Yonghui Wu, and Zhifeng Chen. 2019. GPipe: Efficient Training of Giant Neural Networks using Pipeline Parallelism. In Proceedings of the 33rd International Conference on Neural Information Processing Systems (NIPS), Vol. 32."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2647868.2654889"},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings of the 5th International Conference on Learning Representations (ICLR).","author":"Keskar Nitish Shirish","year":"2017","unstructured":"Nitish Shirish Keskar, Dheevatsa Mudigere, Jorge Nocedal, Mikhail Smelyanskiy, and Ping Tak Peter Tang. 2017. On Large-Batch Training for Deep Learning: Generalization Gap and Sharp Minima. In Proceedings of the 5th International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the 3rd International Conference on Learning Representations (ICLR).","author":"Diederik","unstructured":"Diederik P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. In Proceedings of the 3rd International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.14778\/3342263.3342276"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476145"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.14778\/3415478.3415530"},{"key":"e_1_3_2_1_21_1","volume-title":"An Empirical Model of Large-Batch Training. arXiv","author":"McCandlish Sam","year":"1812","unstructured":"Sam McCandlish, Jared Kaplan, Dario Amodei, and OpenAI Dota Team. 2018. An Empirical Model of Large-Batch Training. arXiv 1812.06162 (2018)."},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the 6th International Conference on Learning Representations (ICLR).","author":"Merity Stephen","year":"2018","unstructured":"Stephen Merity, Nitish Shirish Keskar, and Richard Socher. 2018. Regularizing and Optimizing LSTM Language Models. In Proceedings of the 6th International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning (ICML)","volume":"139","author":"Narayanan Deepak","year":"2021","unstructured":"Deepak Narayanan, Amar Phanishayee, Kaiyu Shi, Xie Chen, and Matei Zaharia. 2021. Memory-Efficient Pipeline-Parallel DNN Training. In Proceedings of the 38th International Conference on Machine Learning (ICML), Vol. 139. 7937--7947."},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of the 40th Annual Meeting on Association for Computational Linguistics (ACL). 311--318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. BLEU: A Method for Automatic Evaluation of Machine Translation. In Proceedings of the 40th Annual Meeting on Association for Computational Linguistics (ACL). 311--318."},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of 2020 USENIX Annual Technical Conference (ATC). 307--321","author":"Park Jay H.","year":"2020","unstructured":"Jay H. Park, Gyeongchan Yun, Chang M. Yi, Nguyen T. Nguyen, Seungmin Lee, Jaesik Choi, Sam H. Noh, and Young ri Choi. 2020. HetPipe: Enabling Large DNN Training on (Whimpy) Heterogeneous GPU Clusters through Integration of Pipelined Model Parallelism and Data Parallelism. In Proceedings of 2020 USENIX Annual Technical Conference (ATC). 307--321."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1137\/0330046"},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the 32nd International Conference on Neural Information Processing Systems (NIPS)","volume":"31","author":"Shazeer Noam","year":"2018","unstructured":"Noam Shazeer, Youlong Cheng, Niki Parmar, Dustin Tran, Ashish Vaswani, Penporn Koanantakool, Peter Hawkins, HyoukJoong Lee, Mingsheng Hong, Cliff Young, Ryan Sepassi, and Blake Hechtman. 2018. Mesh-TensorFlow: Deep Learning for Supercomputers. In Proceedings of the 32nd International Conference on Neural Information Processing Systems (NIPS), Vol. 31."},{"key":"e_1_3_2_1_29_1","volume-title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. arXiv","author":"Shoeybi Mohammad","year":"1909","unstructured":"Mohammad Shoeybi, Mostofa Patwary, Raul Puri, Patrick LeGresley, Jared Casper, and Bryan Catanzaro. 2020. Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. arXiv 1909.08053 (2020)."},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the 7th International Conference on Learning Representations (ICLR).","author":"Stich Sebastian U.","year":"2019","unstructured":"Sebastian U. Stich. 2019. Local SGD Converges Fast and Communicates Little. In Proceedings of the 7th International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the 7th International Conference on Learning Representations (ICLR).","author":"Wang Alex","unstructured":"Alex Wang, Amanpreet Singh, Julian Michael, Felix Hill, Omer Levy, and Samuel R. Bowman. 2019. GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding. In Proceedings of the 7th International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0893-6080(03)00138-2"},{"key":"e_1_3_2_1_33_1","first-page":"269","article-title":"PipeMare: Asynchronous Pipeline Parallel DNN Training","volume":"3","author":"Yang Bowen","year":"2021","unstructured":"Bowen Yang, Jian Zhang, Jonathan Li, Christopher Re, Christopher Aberger, and Christopher De Sa. 2021. PipeMare: Asynchronous Pipeline Parallel DNN Training. In Proceedings of Machine Learning and Systems (MLSys), Vol. 3. 269--296.","journal-title":"Proceedings of Machine Learning and Systems (MLSys)"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.14778\/2732977.2733001"},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the 28th International Conference on Neural Information Processing Systems (NIPS).","author":"Zhang Sixin","year":"2015","unstructured":"Sixin Zhang, Anna Choromanska, and Yann LeCun. 2015. Deep Learning with Elastic Averaging SGD. In Proceedings of the 28th International Conference on Neural Information Processing Systems (NIPS)."}],"event":{"name":"PPoPP '23: The 28th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming","location":"Montreal QC Canada","acronym":"PPoPP '23","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages","SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the 28th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3572848.3577484","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3572848.3577484","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T18:08:09Z","timestamp":1750183689000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3572848.3577484"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,2,21]]},"references-count":35,"alternative-id":["10.1145\/3572848.3577484","10.1145\/3572848"],"URL":"https:\/\/doi.org\/10.1145\/3572848.3577484","relation":{},"subject":[],"published":{"date-parts":[[2023,2,21]]},"assertion":[{"value":"2023-02-21","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}