{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T06:00:17Z","timestamp":1765346417199,"version":"3.46.0"},"reference-count":33,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,5,19]]},"DOI":"10.1109\/ccgrid64434.2025.00044","type":"proceedings-article","created":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T13:36:08Z","timestamp":1751290568000},"page":"472-481","source":"Crossref","is-referenced-by-count":0,"title":["An Optimization Technique for Hiding Communication Costs in 3D Parallel Training of Deep Learning"],"prefix":"10.1109","author":[{"given":"Ryubu","family":"Hosoki","sequence":"first","affiliation":[{"name":"Institute of Science Tokyo,Yokohama,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kento","family":"Sato","sequence":"additional","affiliation":[{"name":"RIKEN Center for Computational Science,Kobe,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Toshio","family":"Endo","sequence":"additional","affiliation":[{"name":"Institute of Science Tokyo,Yokohama,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Julien","family":"Bigot","sequence":"additional","affiliation":[{"name":"Universit&#x00E9; Paris-Saclay, UVSQ, CNRS, CEA, Maison de la Simulation,Gif-sur-Yvette,France"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Edouard","family":"Audit","sequence":"additional","affiliation":[{"name":"Universit&#x00E9; Paris-Saclay, UVSQ, CNRS, CEA, Maison de la Simulation,Gif-sur-Yvette,France"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","author":"Shoeybi","year":"2019","journal-title":"arXiv preprint"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.48550\/arxiv.1811.06965"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"ref5","article-title":"Maximizing Parallelism in Distributed Training for Huge Neural Networks","author":"Bian","year":"2021","journal-title":"arXiv preprint"},{"key":"ref6","first-page":"7937","article-title":"Memory-Efficient Pipeline-Parallel DNN Training","volume-title":"International Conference on Machine Learning. PMLR","author":"Narayanan","year":"2021"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476145"},{"key":"ref8","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3458817.3476209","article-title":"Efficient Large-Scale Language Model Training on GPU Clusters Using Megatron-LM","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis","author":"Narayanan","year":"2021"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406703"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS49936.2021.00109"},{"key":"ref11","first-page":"1","article-title":"Beyond Data and Model Parallelism for Deep Neural Networks","volume-title":"Proceedings of Machine Learning and Systems","volume":"1","author":"Jia","year":"2019"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3132413"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2023.3247001"},{"key":"ref14","first-page":"559","article-title":"Alpa: Automating Inter-and Intra-Operator Parallelism for Distributed Deep Learning","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Zheng","year":"2022"},{"volume-title":"XLA: Compiling Machine Learning for Peak Performance","year":"2020","author":"Sabne","key":"ref15"},{"key":"ref16","article-title":"Using DeepSpeed and Megatron to Train Megatron-Turing NLG 530B, A Large-Scale Generative Language Model","author":"Smith","year":"2022","journal-title":"arXiv preprint"},{"issue":"9","key":"ref17","article-title":"Compiling machine learning programs via high-level tracing","volume":"4","author":"Frostig","year":"2018","journal-title":"Systems for Machine Learning"},{"volume-title":"JAX: composable transformations of Python+NumPy programs","year":"2018","author":"Bradbury","key":"ref18"},{"key":"ref19","first-page":"20","article-title":"Overview of TSUBAME3.0, Green Cloud Supercomputer for Convergence of HPC, AI and Big-Data","volume":"16","author":"Matsuoka","year":"2017","journal-title":"Tsubame ESJ.: e-science journal"},{"key":"ref20","article-title":"GSPMD: General and Scalable Parallelization for ML Computation Graphs","author":"Xu","year":"2021","journal-title":"arXiv preprint"},{"article-title":"Center for Information Infrastructure (CII) of the Institute of Science Tokyo","volume-title":"TSUBAME4 Computing Services","year":"2024","key":"ref21"},{"article-title":"Center for Computational Sciences of University of Tsukuba","volume-title":"Pegasus - Big memory supercomputer","year":"2023","key":"ref22"},{"issue":"8","key":"ref23","first-page":"9","article-title":"Language Models are Unsupervised Multitask Learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI blog"},{"volume-title":"GPT-J-6B: A 6 Billion Parameter Autoregressive Language Model","year":"2021","author":"Wang","key":"ref24"},{"key":"ref25","article-title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","author":"Gu","year":"2023","journal-title":"arXiv preprint"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/iccv48922.2021.00064"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01170"},{"key":"ref28","first-page":"3965","article-title":"Coatnet: Marrying Convolution and Attention for All Data Sizes","volume":"34","author":"Dai","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.14618\/IDS-PUB-9021"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.156"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"volume-title":"ONNX: Open Neural Network Exchange","author":"Bai","key":"ref32"},{"key":"ref33","article-title":"OneFlow: Redesign the Distributed Deep Learning Framework from Scratch","author":"Yuan","year":"2021","journal-title":"arXiv preprint"}],"event":{"name":"2025 IEEE 25th International Symposium on Cluster, Cloud and Internet Computing (CCGrid)","start":{"date-parts":[[2025,5,19]]},"location":"Troms\u00f8, Norway","end":{"date-parts":[[2025,5,22]]}},"container-title":["2025 IEEE 25th International Symposium on Cluster, Cloud and Internet Computing (CCGrid)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11044421\/11044790\/11044826.pdf?arnumber=11044826","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:56:49Z","timestamp":1765346209000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11044826\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,19]]},"references-count":33,"URL":"https:\/\/doi.org\/10.1109\/ccgrid64434.2025.00044","relation":{},"subject":[],"published":{"date-parts":[[2025,5,19]]}}}