{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,2]],"date-time":"2025-11-02T15:11:44Z","timestamp":1762096304322,"version":"build-2065373602"},"publisher-location":"New York, NY, USA","reference-count":40,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,11,12]],"date-time":"2023-11-12T00:00:00Z","timestamp":1699747200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,11,12]]},"DOI":"10.1145\/3624062.3626080","type":"proceedings-article","created":{"date-parts":[[2023,11,10]],"date-time":"2023-11-10T13:53:39Z","timestamp":1699624419000},"page":"44-50","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Elastic deep learning through resilient collective operations"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7929-8484","authenticated-orcid":false,"given":"Jiali","family":"Li","sequence":"first","affiliation":[{"name":"the University of Tennessee, Knoxville, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2411-8495","authenticated-orcid":false,"given":"George","family":"Bosilca","sequence":"additional","affiliation":[{"name":"the University of Tennessee, Knoxville, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5108-509X","authenticated-orcid":false,"given":"Aurelien","family":"Bouteiller","sequence":"additional","affiliation":[{"name":"the University of Tennessee, Knoxville, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0661-7509","authenticated-orcid":false,"given":"Bogdan","family":"Nicolae","sequence":"additional","affiliation":[{"name":"Argonne National Laboratory (ANL), United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,11,12]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"[n. d.]. Gloo. https:\/\/github.com\/facebookincubator\/gloo Last accessed on 2023-7-4."},{"key":"e_1_3_2_2_2_1","unstructured":"[n. d.]. NCCL. https:\/\/github.com\/NVIDIA\/nccl Last accessed on 2023-7-4."},{"key":"e_1_3_2_2_3_1","unstructured":"2019. Horoovd. https:\/\/github.com\/horovod\/horovod Last accessed on 2023-7-4."},{"key":"e_1_3_2_2_4_1","unstructured":"2019. Pytorch elastic. https:\/\/github.com\/pytorch\/elastic Last accessed on 2023-7-4."},{"key":"e_1_3_2_2_5_1","unstructured":"2022. Kaggle Fruits 360 datasets. https:\/\/www.kaggle.com\/datasets\/moltean\/fruits"},{"key":"e_1_3_2_2_6_1","unstructured":"2022. Keras Applications. https:\/\/keras.io\/api\/applications\/"},{"key":"e_1_3_2_2_7_1","volume-title":"Replication Based Fault Tolerance Approach for Cloud. In International Conference on Distributed Computing and Internet Technology. Springer, 163\u2013169","author":"Agarwal K","year":"2022","unstructured":"Kamal\u00a0K Agarwal and Haribabu Kotakula. 2022. Replication Based Fault Tolerance Approach for Cloud. In International Conference on Distributed Computing and Internet Technology. Springer, 163\u2013169."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-67077-1_2"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3127024.3127037"},{"key":"e_1_3_2_2_10_1","volume-title":"Implicit Actions and Non-blocking Failure Recovery with MPI. In 2022 IEEE\/ACM 12th Workshop on Fault Tolerance for HPC at eXtreme Scale (FTXS). IEEE, 36\u201346","author":"Bouteiller Aurelien","year":"2022","unstructured":"Aurelien Bouteiller and George Bosilca. 2022. Implicit Actions and Non-blocking Failure Recovery with MPI. In 2022 IEEE\/ACM 12th Workshop on Fault Tolerance for HPC at eXtreme Scale (FTXS). IEEE, 36\u201346."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342006067469"},{"key":"e_1_3_2_2_12_1","volume-title":"An Algorithm-Based Fault Tolerance Strategy for the Bitonic Sort Parallel Algorithm. In 2021 10th Latin-American Symposium on Dependable Computing (LADC). IEEE, 1\u201310","author":"Camargo T","year":"2021","unstructured":"Edson\u00a0T Camargo and Elias\u00a0P Duarte. 2021. An Algorithm-Based Fault Tolerance Strategy for the Bitonic Sort Parallel Algorithm. In 2021 10th Latin-American Symposium on Dependable Computing (LADC). IEEE, 1\u201310."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421307"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3436728"},{"key":"e_1_3_2_2_15_1","volume-title":"Fault-tolerant control of an error-corrected qubit. Nature 598, 7880","author":"Egan Laird","year":"2021","unstructured":"Laird Egan, Dripto\u00a0M Debroy, Crystal Noel, Andrew Risinger, Daiwei Zhu, Debopriyo Biswas, Michael Newman, Muyuan Li, Kenneth\u00a0R Brown, Marko Cetina, 2021. Fault-tolerant control of an error-corrected qubit. Nature 598, 7880 (2021), 281\u2013286."},{"key":"e_1_3_2_2_16_1","volume-title":"large minibatch sgd: Training imagenet in 1 hour. arXiv preprint arXiv:1706.02677","author":"Goyal Priya","year":"2017","unstructured":"Priya Goyal, Piotr Doll\u00e1r, Ross Girshick, Pieter Noordhuis, Lukasz Wesolowski, Aapo Kyrola, Andrew Tulloch, Yangqing Jia, and Kaiming He. 2017. Accurate, large minibatch sgd: Training imagenet in 1 hour. arXiv preprint arXiv:1706.02677 (2017)."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3332372"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/2931028.2931030"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/TDSC.2021.3063083"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1002\/spe.3021"},{"key":"e_1_3_2_2_21_1","unstructured":"J.Stengler. 2017. Fault tolerant Collective communication Algorithms For distributed database systems. (2017)."},{"key":"e_1_3_2_2_22_1","volume-title":"One weird trick for parallelizing convolutional neural networks. arXiv preprint arXiv:1404.5997","author":"Krizhevsky Alex","year":"2014","unstructured":"Alex Krizhevsky. 2014. One weird trick for parallelizing convolutional neural networks. arXiv preprint arXiv:1404.5997 (2014)."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS.2019.00203"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.future.2020.01.026"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2013.14"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid49817.2020.00-76"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER49012.2020.00033"},{"key":"e_1_3_2_2_28_1","first-page":"400","article-title":"Resource elasticity in distributed deep learning","volume":"2","author":"Or Andrew","year":"2020","unstructured":"Andrew Or, Haoyu Zhang, and Michael Freedman. 2020. Resource elasticity in distributed deep learning. Proceedings of Machine Learning and Systems 2 (2020), 400\u2013411.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_2_29_1","volume-title":"2018 USENIX Annual Technical Conference (USENIX ATC 18)","author":"Qiao Aurick","year":"2018","unstructured":"Aurick Qiao, Abutalib Aghayev, Weiren Yu, Haoyang Chen, Qirong Ho, Garth\u00a0A Gibson, and Eric\u00a0P Xing. 2018. Litz: Elastic framework for { High-Performance} distributed machine learning. In 2018 USENIX Annual Technical Conference (USENIX ATC 18). 631\u2013644."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/AERO47225.2020.9172799"},{"key":"e_1_3_2_2_31_1","volume-title":"A study of checkpointing in large scale training of deep neural networks. arXiv preprint arXiv:2012.00825","author":"Rojas Elvis","year":"2020","unstructured":"Elvis Rojas, Albert\u00a0Njoroge Kahira, Esteban Meneses, Leonardo\u00a0Bautista Gomez, and Rosa\u00a0M Badia. 2020. A study of checkpointing in large scale training of deep neural networks. arXiv preprint arXiv:2012.00825 (2020)."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2019.09.004"},{"key":"e_1_3_2_2_33_1","first-page":"475","article-title":"An Efficient Data Replication Technique with Fault Tolerance Approach using BVAG with Checkpoint and Rollback-Recovery","volume":"12","author":"Sy\u00a0Ahmad Ubaidillah Sharifah Hafizah","year":"2021","unstructured":"Sharifah Hafizah Sy\u00a0Ahmad Ubaidillah, Basem Alkazemi, and A Noraziah. 2021. An Efficient Data Replication Technique with Fault Tolerance Approach using BVAG with Checkpoint and Rollback-Recovery. International Journal of Advanced Computer Science and Applications 12, 1 (2021), 475\u2013480.","journal-title":"International Journal of Advanced Computer Science and Applications"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER51413.2022.00052"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3064966"},{"key":"e_1_3_2_2_36_1","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18)","author":"Xiao Wencong","year":"2018","unstructured":"Wencong Xiao, Romil Bhardwaj, Ramachandran Ramjee, Muthian Sivathanu, Nipun Kwatra, Zhenhua Han, Pratyush Patel, Xuan Peng, Hanyu Zhao, Quanlu Zhang, 2018. Gandiva: Introspective cluster scheduling for deep learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18). 595\u2013610."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS47774.2020.00018"},{"key":"e_1_3_2_2_38_1","first-page":"1677","article-title":"FT-CNN: Algorithm-based fault tolerance for convolutional neural networks","volume":"32","author":"Zhao Kai","year":"2020","unstructured":"Kai Zhao, Sheng Di, Sihuan Li, Xin Liang, Yujia Zhai, Jieyang Chen, Kaiming Ouyang, Franck Cappello, and Zizhong Chen. 2020. FT-CNN: Algorithm-based fault tolerance for convolutional neural networks. IEEE Transactions on Parallel and Distributed Systems 32, 7 (2020), 1677\u20131689.","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"e_1_3_2_2_39_1","volume-title":"Hycor: Fault-tolerant replicated containers based on checkpoint and replay. arXiv preprint arXiv:2101.09584","author":"Zhou Diyu","year":"2021","unstructured":"Diyu Zhou and Yuval Tamir. 2021. Hycor: Fault-tolerant replicated containers based on checkpoint and replay. arXiv preprint arXiv:2101.09584 (2021)."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539597.3573037"}],"event":{"name":"SC-W 2023: Workshops of The International Conference on High Performance Computing, Network, Storage, and Analysis","acronym":"SC-W 2023","location":"Denver CO USA"},"container-title":["Proceedings of the SC '23 Workshops of the International Conference on High Performance Computing, Network, Storage, and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3624062.3626080","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3624062.3626080","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T03:04:05Z","timestamp":1755745445000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3624062.3626080"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,12]]},"references-count":40,"alternative-id":["10.1145\/3624062.3626080","10.1145\/3624062"],"URL":"https:\/\/doi.org\/10.1145\/3624062.3626080","relation":{},"subject":[],"published":{"date-parts":[[2023,11,12]]},"assertion":[{"value":"2023-11-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}