{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T21:51:09Z","timestamp":1781905869276,"version":"3.54.5"},"reference-count":53,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2025,4,21]],"date-time":"2025-04-21T00:00:00Z","timestamp":1745193600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,4,21]],"date-time":"2025-04-21T00:00:00Z","timestamp":1745193600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2025,5]]},"DOI":"10.1007\/s11432-022-4201-7","type":"journal-article","created":{"date-parts":[[2025,4,23]],"date-time":"2025-04-23T23:28:05Z","timestamp":1745450885000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["FMCC-RT: a scalable and fine-grained all-reduce algorithm for large-scale SMP clusters"],"prefix":"10.1007","volume":"68","author":[{"given":"Jintao","family":"Peng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jie","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianbin","family":"Fang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Min","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yi","family":"Dai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhiquan","family":"Lai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bo","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chunye","family":"Gong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinjun","family":"Mao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guo","family":"Mao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jie","family":"Ren","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,4,21]]},"reference":[{"key":"4201_CR1","first-page":"626","volume-title":"Proceedings of IEEE Conference on Computer Communications","author":"Y Bao","year":"2020","unstructured":"Bao Y, Peng Y, Chen Y, et al. Preemptive all-reduce scheduling for expediting distributed DNN training. In: Proceedings of IEEE Conference on Computer Communications, 2020. 626\u2013635"},{"key":"4201_CR2","first-page":"172","volume-title":"Proceedings of IEEE Conference on Computer Communications","author":"S Shi","year":"2019","unstructured":"Shi S, Chu X, Li B. MG-WFBP: efficient data communication for distributed synchronous SGD algorithms. In: Proceedings of IEEE Conference on Computer Communications, 2019. 172\u2013180"},{"key":"4201_CR3","first-page":"994","volume-title":"Proceedings of IEEE International Parallel and Distributed Processing Symposium (IPDPS)","author":"Y Ko","year":"2021","unstructured":"Ko Y, Choi K, Seo J, et al. An in-depth analysis of distributed training of deep neural networks. In: Proceedings of IEEE International Parallel and Distributed Processing Symposium (IPDPS), 2021. 994\u20131003"},{"key":"4201_CR4","first-page":"386","volume-title":"Proceedings of International Conference for High Performance Computing, Networking, Storage and Analysis","author":"S Chunduri","year":"2018","unstructured":"Chunduri S, Parker S, Balaji P, et al. Characterization of MPI usage on a production supercomputer. In: Proceedings of International Conference for High Performance Computing, Networking, Storage and Analysis, 2018. 386\u2013400"},{"key":"4201_CR5","doi-asserted-by":"publisher","first-page":"117","DOI":"10.1016\/j.jpdc.2008.09.002","volume":"69","author":"P Patarasuk","year":"2009","unstructured":"Patarasuk P, Yuan X. Bandwidth optimal all-reduce algorithms for clusters of workstations. J Parallel Distr Comput, 2009, 69: 117\u2013124","journal-title":"J Parallel Distr Comput"},{"key":"4201_CR6","first-page":"1031","volume-title":"Proceedings of IEEE International Parallel and Distributed Processing Symposium (IPDPS)","author":"S Kumar","year":"2016","unstructured":"Kumar S, Sharkawi S S, Jan K N. Optimization and analysis of MPI collective communication on fat-tree networks. In: Proceedings of IEEE International Parallel and Distributed Processing Symposium (IPDPS), 2016. 1031\u20131040"},{"key":"4201_CR7","doi-asserted-by":"publisher","first-page":"389","DOI":"10.1016\/S0167-8191(06)80021-9","volume":"20","author":"R W Hockney","year":"1994","unstructured":"Hockney R W. The communication challenge for MPP: Intel Paragon and Meiko CS-2. Parallel Comput, 1994, 20: 389\u2013398","journal-title":"Parallel Comput"},{"key":"4201_CR8","first-page":"1","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis","author":"M Bayatpour","year":"2017","unstructured":"Bayatpour M, Chakraborty S, Subramoni H, et al. Scalable reduction collectives with data partitioning-based multi-core design. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, 2017. 1\u201311"},{"key":"4201_CR9","doi-asserted-by":"publisher","first-page":"e5574","DOI":"10.1002\/cpe.5574","volume":"33","author":"T T Nguyen","year":"2021","unstructured":"Nguyen T T, Wahib M, Takano R. Efficient MPI-AllReduce for large-scale deep learning on GPU-clusters. Concurr Comput, 2021, 33: e5574","journal-title":"Concurr Comput"},{"key":"4201_CR10","first-page":"430","volume-title":"Proceedings of the 19th IEEE\/ACM International Symposium on Cluster, Cloud and Grid Computing (CCGRID)","author":"Y Ueno","year":"2019","unstructured":"Ueno Y, Yokota R. Exhaustive study of hierarchical AllReduce patterns for large messages between GPUs. In: Proceedings of the 19th IEEE\/ACM International Symposium on Cluster, Cloud and Grid Computing (CCGRID), 2019. 430\u2013439"},{"key":"4201_CR11","first-page":"1","volume-title":"Proceedings of International Conference on Computational Science","author":"R Rabenseifner","year":"2004","unstructured":"Rabenseifner R. Optimization of collective reduction operations. In: Proceedings of International Conference on Computational Science, 2004. 1\u20139"},{"key":"4201_CR12","first-page":"216","volume-title":"Proceedings of the 6th International Symposium on Computing and Networking Workshops (CANDARW)","author":"T T Nguyen","year":"2018","unstructured":"Nguyen T T, Wahib M, Takano R. Hierarchical distributed-memory multi-leader MPI-Allreduce for deep learning workloads. In: Proceedings of the 6th International Symposium on Computing and Networking Workshops (CANDARW), 2018. 216\u2013222"},{"key":"4201_CR13","first-page":"270","volume-title":"Proceedings of IEEE International Conference on Cluster Computing (CLUSTER)","author":"J L Tr\u00e4ff","year":"2020","unstructured":"Tr\u00e4ff J L, Hunold S. Decomposing MPI collectives for exploiting multi-lane communication. In: Proceedings of IEEE International Conference on Cluster Computing (CLUSTER), 2020. 270\u2013280"},{"key":"4201_CR14","first-page":"402","volume-title":"Proceedings of IEEE International Conference on Parallel & Distributed Processing with Applications, Big Data & Cloud Computing, Sustainable Computing & Communications, Social Computing & Networking (ISPA\/BDCloud\/SocialCom\/SustainCom)","author":"J Peng","year":"2022","unstructured":"Peng J, Liu J, Dai Y, et al. Optimizing all-to-all collective communication on Tianhe supercomputer. In: Proceedings of IEEE International Conference on Parallel & Distributed Processing with Applications, Big Data & Cloud Computing, Sustainable Computing & Communications, Social Computing & Networking (ISPA\/BDCloud\/SocialCom\/SustainCom), 2022. 402\u2013409"},{"key":"4201_CR15","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1109\/MCOM.2006.1668378","volume":"44","author":"S A Reinemo","year":"2006","unstructured":"Reinemo S A, Skeie T, Sodring T, et al. An overview of QoS capabilities in infiniband, advanced switching interconnect, and ethernet. IEEE Commun Mag, 2006, 44: 32\u201338","journal-title":"IEEE Commun Mag"},{"key":"4201_CR16","first-page":"1","volume-title":"Proceedings of IEEE 24th Annual Symposium on High-Performance Interconnects (HOTI)","author":"T Schneider","year":"2016","unstructured":"Schneider T, Bibartiu O, Hoefler T. Ensuring deadlock-freedom in low-diameter infiniband networks. In: Proceedings of IEEE 24th Annual Symposium on High-Performance Interconnects (HOTI), 2016. 1\u20138"},{"key":"4201_CR17","volume-title":"MVAPICH: MPI over InfiniBand, Omni-Path, Ethernet\/iWARP, RoCE, and Slingshot","author":"Laboratory O N B C","year":"2022","unstructured":"Laboratory O N B C. MVAPICH: MPI over InfiniBand, Omni-Path, Ethernet\/iWARP, RoCE, and Slingshot. 2022. https:\/\/mvapich.cse.ohio-state.edu\/benchmarks\/"},{"key":"4201_CR18","first-page":"871","volume-title":"Proceedings of IEEE International Parallel and Distributed Processing Symposium","author":"R Belli","year":"2015","unstructured":"Belli R, Hoefler T. Notified access: extending remote memory access programming models for producer-consumer synchronization. In: Proceedings of IEEE International Parallel and Distributed Processing Symposium, 2015. 871\u2013881"},{"key":"4201_CR19","first-page":"1072","volume-title":"Proceedings of the 24th Annual Joint Conference of the IEEE Computer and Communications Societies","author":"A Dhamdhere","year":"2005","unstructured":"Dhamdhere A, Jiang H, Dovrolis C. Buffer sizing for congested Internet links. In: Proceedings of the 24th Annual Joint Conference of the IEEE Computer and Communications Societies, 2005. 1072\u20131083"},{"key":"4201_CR20","doi-asserted-by":"publisher","first-page":"169","DOI":"10.1016\/j.comnet.2004.04.002","volume":"46","author":"A E Kamal","year":"2004","unstructured":"Kamal A E, Hassanein H S. Performance evaluation of prioritized scheduling with buffer management for differentiated services architectures. Comput Netws, 2004, 46: 169\u2013180","journal-title":"Comput Netws"},{"key":"4201_CR21","volume-title":"Proceedings of International Parallel and Distributed Processing Symposium","author":"J C Sancho","year":"2002","unstructured":"Sancho J C, Flich J, Robles A, et al. Analyzing the influence of virtual lanes on the performance of infiniband networks. In: Proceedings of International Parallel and Distributed Processing Symposium, 2002"},{"key":"4201_CR22","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1145\/3503221.3508399","volume-title":"Proceedings of the 27th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","author":"S Li","year":"2022","unstructured":"Li S, Hoefler T. Near-optimal sparse allreduce for distributed deep learning. In: Proceedings of the 27th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, 2022. 135\u2013149"},{"key":"4201_CR23","doi-asserted-by":"publisher","first-page":"62","DOI":"10.1145\/3437801.3441620","volume-title":"Proceedings of the 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","author":"Z Cai","year":"2021","unstructured":"Cai Z, Liu Z, Maleki S, et al. Synthesizing optimal collective algorithms. In: Proceedings of the 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, 2021. 62\u201375"},{"key":"4201_CR24","volume-title":"Proceedings of the 2nd SysML Conference","author":"M Cho","year":"2019","unstructured":"Cho M, Finkler U, Kung D. Blueconnect: novel hierarchical all-reduce on multi-tired network for deep learning. In: Proceedings of the 2nd SysML Conference, 2019"},{"key":"4201_CR25","volume-title":"Massively scale your deep learning training with NCCL 2.4","author":"Mellanox","year":"2022","unstructured":"Mellanox. Massively scale your deep learning training with NCCL 2.4. 2022. https:\/\/developer.nvidia.com"},{"key":"4201_CR26","first-page":"463","volume-title":"Proceedings of the 14th USENIX Conference on Operating Systems Design and Implementation","author":"Y Jiang","year":"2020","unstructured":"Jiang Y, Zhu Y, Lan C, et al. A unified architecture for accelerating distributed DNN training in heterogeneous GPU\/CPU clusters. In: Proceedings of the 14th USENIX Conference on Operating Systems Design and Implementation, 2020. 463\u2013479"},{"key":"4201_CR27","first-page":"1","volume-title":"Proceedings of the ACM\/IEEE conference on Supercomputing","author":"T Hoefler","year":"2007","unstructured":"Hoefler T, Lumsdaine A, Rehm W. Implementation and performance analysis of non-blocking collective operations for MPI. In: Proceedings of the ACM\/IEEE conference on Supercomputing, 2007. 1\u201310"},{"key":"4201_CR28","first-page":"12","volume-title":"Proceedings of IEEE International Conference on Cluster Computing (CLUSTER)","author":"M Bayatpour","year":"2018","unstructured":"Bayatpour M, Hashmi J M, Chakraborty S, et al. Salar: scalable and adaptive designs for large message reduction collectives. In: Proceedings of IEEE International Conference on Cluster Computing (CLUSTER), 2018. 12\u201323"},{"key":"4201_CR29","doi-asserted-by":"publisher","first-page":"63","DOI":"10.1109\/TCC.2015.2415804","volume":"4","author":"E Yildirim","year":"2015","unstructured":"Yildirim E, Arslan E, Kim J, et al. Application-level optimization of big data transfers through pipelining, parallelism and concurrency. IEEE Trans Cloud Comput, 2015, 4: 63\u201375","journal-title":"IEEE Trans Cloud Comput"},{"key":"4201_CR30","first-page":"1","volume-title":"Proceedings of IEEE International Symposium on Parallel & Distributed Processing, Workshops and PhD Forum (IPDPSW)","author":"H Kamal","year":"2010","unstructured":"Kamal H, Wagner A. FG-MPI: fine-grain MPI for multicore and clusters. In: Proceedings of IEEE International Symposium on Parallel & Distributed Processing, Workshops and PhD Forum (IPDPSW), 2010. 1\u20138"},{"key":"4201_CR31","unstructured":"Jia X, Song S, He W, et al. Highly scalable deep learning training system with mixed-precision: training ImageNet in four minutes. 2018. ArXiv:1807.11205"},{"key":"4201_CR32","unstructured":"Dunkels A. Protothreads. 2022. http:\/\/dunkels.com\/adam\/pt\/index.html"},{"key":"4201_CR33","doi-asserted-by":"publisher","first-page":"345","DOI":"10.1007\/s11704-014-3501-3","volume":"8","author":"X Liao","year":"2014","unstructured":"Liao X, Xiao L, Yang C, et al. MilkyWay-2 supercomputer: system and application. Front Comput Sci, 2014, 8: 345\u2013356","journal-title":"Front Comput Sci"},{"key":"4201_CR34","doi-asserted-by":"publisher","first-page":"150","DOI":"10.1007\/s42514-022-00095-y","volume":"4","author":"K Lu","year":"2022","unstructured":"Lu K, Wang Y, Guo Y, et al. MT-3000: a heterogeneous multi-zone processor for HPC. CCF Trans HPC, 2022, 4: 150\u2013164","journal-title":"CCF Trans HPC"},{"key":"4201_CR35","doi-asserted-by":"publisher","first-page":"509","DOI":"10.1631\/FITEE.2200359","volume":"24","author":"J Fang","year":"2023","unstructured":"Fang J, Zhang P, Huang C, et al. Programming bare-metal accelerators with heterogeneous threading models: a case study of Matrix-3000. Front Inform Technol Electron Eng, 2023, 24: 509\u2013520","journal-title":"Front Inform Technol Electron Eng"},{"key":"4201_CR36","volume-title":"Performance reported by NCCL tests","author":"Nvidia","year":"2022","unstructured":"Nvidia. Performance reported by NCCL tests. 2022. https:\/\/github.com\/NVIDIA\/nccl-tests\/blob\/master\/doc\/PERFORMANCE.md"},{"key":"4201_CR37","first-page":"593","volume-title":"Proceedings of the 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"A Shah","year":"2023","unstructured":"Shah A, Chidambaram V, Cowan M, et al. TACCL: guiding collective algorithm synthesis using communication sketches. In: Proceedings of the 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23), 2023. 593\u2013612"},{"key":"4201_CR38","unstructured":"Sergeev A, Balso M D. Horovod: fast and easy distributed deep learning in TensorFlow. 2018. ArXiv:1802.05799"},{"key":"4201_CR39","unstructured":"Horovod. Horovod. 2023. https:\/\/github.com\/horovod\/horovod"},{"key":"4201_CR40","first-page":"770","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"K He","year":"2016","unstructured":"He K, Zhang X, Ren S, et al. Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016. 770\u2013778"},{"key":"4201_CR41","unstructured":"Simonyan K, Zisserman A. Very deep convolutional networks for large-scale image recognition. 2014. ArXiv:1409.1556"},{"key":"4201_CR42","series-title":"Technical Report","volume-title":"Miniamr\u2014A Miniapp For Adaptive Mesh Refinement","author":"A Sasidharan","year":"2016","unstructured":"Sasidharan A, Snir M. Miniamr\u2014A Miniapp For Adaptive Mesh Refinement. Technical Report, 2016"},{"key":"4201_CR43","doi-asserted-by":"publisher","first-page":"118","DOI":"10.1007\/978-3-030-78713-4_7","volume-title":"Proceedings of 36th International Conference on High Performance Computing","author":"K K Shafie","year":"2021","unstructured":"Shafie K K, Hashmi J, Chu C H, et al. Designing a ROCm-aware MPI library for AMD GPUs: early experiences. In: Proceedings of 36th International Conference on High Performance Computing, 2021. 118\u2013136"},{"key":"4201_CR44","unstructured":"Intel. Intel oneCCL. 2023. https:\/\/github.com\/intel\/torch-ccl"},{"key":"4201_CR45","unstructured":"Ren X, Zhou P, Meng X, et al. PanGu-\u03a3: towards trillion parameter language model with sparse heterogeneous computing. 2023. ArXiv:2303.10845"},{"key":"4201_CR46","first-page":"33","volume-title":"Proceedings of IEEE\/ACM International Workshop on Heterogeneous High-performance Reconfigurable Computing (H2RC)","author":"Z He","year":"2021","unstructured":"He Z, Parravicini D, Petrica L, et al. ACCL: FPGA-accelerated collectives over 100 Gbps TCP-IP. In: Proceedings of IEEE\/ACM International Workshop on Heterogeneous High-performance Reconfigurable Computing (H2RC), 2021. 33\u201343"},{"key":"4201_CR47","unstructured":"Huawei. Tensorflow network model porting and training guide. https:\/\/support.huawei.com\/enterprise\/en\/doc\/EDOC1100164821\/74dd77b\/distributed-training-based-on-the-allreduce-architecture"},{"key":"4201_CR48","unstructured":"Microsoft. Microsoft collective communication library (MSCCL). 2023. https:\/\/github.com\/microsoft\/msccl"},{"key":"4201_CR49","unstructured":"facebookincubator. Gloo. 2023. https:\/\/github.com\/facebookincubator\/gloo"},{"key":"4201_CR50","unstructured":"MPI O. Open MPI. 2023. https:\/\/github.com\/open-mpi\/ompi"},{"key":"4201_CR51","doi-asserted-by":"publisher","first-page":"49","DOI":"10.1177\/1094342005051521","volume":"19","author":"R Thakur","year":"2005","unstructured":"Thakur R, Rabenseifner R, Gropp W. Optimization of collective communication operations in MPICH. Int J High Performance Computing Appl, 2005, 19: 49\u201366","journal-title":"Int J High Performance Computing Appl"},{"key":"4201_CR52","volume-title":"Proceedings of MVAPICH2 User Group (MUG) Meeting","author":"D Panda","year":"2019","unstructured":"Panda D. Overview of the MVAPICH project: latest status and future roadmap. In: Proceedings of MVAPICH2 User Group (MUG) Meeting, 2019"},{"key":"4201_CR53","unstructured":"Shi S, Chu X, Cheung K C, et al. Understanding top-k sparsification in distributed deep learning. 2019. ArXiv:1911.08772"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-022-4201-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-022-4201-7","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-022-4201-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T21:02:44Z","timestamp":1781902964000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-022-4201-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,21]]},"references-count":53,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2025,5]]}},"alternative-id":["4201"],"URL":"https:\/\/doi.org\/10.1007\/s11432-022-4201-7","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,4,21]]},"assertion":[{"value":"4 August 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 April 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 November 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 April 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"152103"}}