{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T02:03:37Z","timestamp":1780020217692,"version":"3.53.1"},"reference-count":49,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100003787","name":"Hebei Provincial Natural Science Foundation","doi-asserted-by":"publisher","award":["F2025525008"],"award-info":[{"award-number":["F2025525008"]}],"id":[{"id":"10.13039\/501100003787","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Future Generation Computer Systems"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.future.2026.108498","type":"journal-article","created":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T03:46:07Z","timestamp":1774583167000},"page":"108498","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Maximizing the computation-communication overlap for distributed deep learning with approximate AllReduce"],"prefix":"10.1016","volume":"182","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4041-3681","authenticated-orcid":false,"given":"Shouxi","family":"Luo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gaolin","family":"Tang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xue","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6345-7265","authenticated-orcid":false,"given":"Huanlai","family":"Xing","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"2","key":"10.1016\/j.future.2026.108498_bib0001","doi-asserted-by":"crossref","DOI":"10.1145\/3377454","article-title":"A survey on distributed machine learning","volume":"53","author":"Verbraeken","year":"2020","journal-title":"ACM Comput. Surv."},{"issue":"1","key":"10.1016\/j.future.2026.108498_bib0002","doi-asserted-by":"crossref","DOI":"10.1155\/int\/3715086","article-title":"A resilience recovery method for complex traffic network security based on trend forecasting","volume":"2025","author":"Hong","year":"2025","journal-title":"Int. J. Intell. Syst."},{"key":"10.1016\/j.future.2026.108498_bib0003","doi-asserted-by":"crossref","DOI":"10.1016\/j.future.2025.107983","article-title":"Approximate gradient synchronization with adaptive quantized gradient broadcast","volume":"174","author":"Luo","year":"2026","journal-title":"Future Generat. Comput. Syst."},{"issue":"5","key":"10.1016\/j.future.2026.108498_bib0004","doi-asserted-by":"crossref","first-page":"4793","DOI":"10.1109\/TNSE.2024.3419030","article-title":"Efficient inter-datacenter AllReduce with multiple trees","volume":"11","author":"Luo","year":"2024","journal-title":"IEEE Trans. Netw. Sci. Eng."},{"issue":"5","key":"10.1016\/j.future.2026.108498_bib0005","doi-asserted-by":"crossref","first-page":"4488","DOI":"10.1109\/TNET.2024.3423380","article-title":"Releasing the power of in-network aggregation with aggregator-aware routing optimization","volume":"32","author":"Luo","year":"2024","journal-title":"IEEE\/ACM Trans. Network."},{"key":"10.1016\/j.future.2026.108498_bib0006","series-title":"2025 IEEE\/CIC International Conference on Communications in China (ICCC)","first-page":"1","article-title":"Pushing the performance boundary of in-network AllReduce with joint topology and routing optimization","author":"Luo","year":"2025"},{"key":"10.1016\/j.future.2026.108498_bib0007","series-title":"2025 IEEE\/CIC International Conference on Communications in China (ICCC)","first-page":"1","article-title":"Maximizing the throughput of edge-based in-network aggregation with routing optimization","author":"Qiao","year":"2025"},{"key":"10.1016\/j.future.2026.108498_bib0008","article-title":"Communication efficient distributed machine learning with the parameter server","volume":"27","author":"Li","year":"2014","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.future.2026.108498_bib0009","series-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","first-page":"559","article-title":"Alpa: automating inter-and intra-operator parallelism for distributed deep learning","author":"Zheng","year":"2022"},{"key":"10.1016\/j.future.2026.108498_bib0010","series-title":"Proceedings of the 27th ACM Symposium on Operating Systems Principles","first-page":"16","article-title":"A generic communication scheduler for distributed DNN training acceleration","author":"Peng","year":"2019"},{"key":"10.1016\/j.future.2026.108498_bib0011","series-title":"2021 IEEE 41st International Conference on Distributed Computing Systems (ICDCS)","first-page":"561","article-title":"GRACE: A compressed communication framework for distributed machine learning","author":"Xu","year":"2021"},{"key":"10.1016\/j.future.2026.108498_bib0012","series-title":"International Conference on Machine Learning","first-page":"5325","article-title":"Error compensated quantized SGD and its applications to large-scale distributed optimization","author":"Wu","year":"2018"},{"issue":"9","key":"10.1016\/j.future.2026.108498_bib0013","doi-asserted-by":"crossref","first-page":"2144","DOI":"10.1109\/TPDS.2021.3062721","article-title":"Overlapping communication with computation in parameter server for scalable DL training","volume":"32","author":"Wang","year":"2021","journal-title":"IEEE Trans. Parall. Distrib. Syst."},{"issue":"3","key":"10.1016\/j.future.2026.108498_bib0014","doi-asserted-by":"crossref","first-page":"230","DOI":"10.1109\/MNET.011.2000530","article-title":"A quantitative survey of communication optimizations in distributed deep learning","volume":"35","author":"Shi","year":"2020","journal-title":"IEEE Netw."},{"key":"10.1016\/j.future.2026.108498_bib0015","series-title":"Proceedings of the 6th Asia-Pacific Workshop on Networking, APNet\u201922","first-page":"101","article-title":"Approximate gradient synchronization with AQGB","author":"Liu","year":"2023"},{"issue":"1","key":"10.1016\/j.future.2026.108498_bib0016","doi-asserted-by":"crossref","first-page":"49","DOI":"10.1177\/1094342005051521","article-title":"Optimization of collective communication operations in MPICH","volume":"19","author":"Thakur","year":"2005","journal-title":"Int. J. High Perform. Comput. Appl."},{"issue":"12","key":"10.1016\/j.future.2026.108498_bib0017","doi-asserted-by":"crossref","DOI":"10.1002\/cpe.5574","article-title":"Efficient MPI-AllReduce for large-scale deep learning on GPU-clusters","volume":"33","author":"Thao Nguyen","year":"2021","journal-title":"Concurr. Comput.: Pract. Exper."},{"issue":"11","key":"10.1016\/j.future.2026.108498_bib0018","doi-asserted-by":"crossref","first-page":"2224","DOI":"10.1109\/TPDS.2024.3460185","article-title":"Efficient cross-cloud partial reduce with CREW","volume":"35","author":"Luo","year":"2024","journal-title":"IEEE Trans. Parall. Distrib. Syst."},{"key":"10.1016\/j.future.2026.108498_bib0019","first-page":"1","article-title":"Efficient parameter synchronization for peer-to-peer distributed learning with selective multicast","author":"Luo","year":"2024","journal-title":"IEEE Trans. Serv. Comput."},{"key":"10.1016\/j.future.2026.108498_bib0020","series-title":"Proceedings of the 14th USENIX Conference on Operating Systems Design and Implementation, OSDI\u201920","article-title":"KungFu: making training in distributed machine learning adaptive","author":"Mai","year":"2020"},{"key":"10.1016\/j.future.2026.108498_bib0021","series-title":"11th USENIX Symposium on Operating Systems Design and Implementation (OSDI 14)","first-page":"583","article-title":"Scaling distributed machine learning with the parameter server","author":"Li","year":"2014"},{"key":"10.1016\/j.future.2026.108498_bib0022","series-title":"ICC 2022 - IEEE International Conference on Communications","first-page":"4775","article-title":"Fast parameter synchronization for distributed learning with selective multicast","author":"Luo","year":"2022"},{"key":"10.1016\/j.future.2026.108498_bib0023","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2014-274","article-title":"1-bit stochastic gradient descent and its application to data-parallel distributed training of speech DNNs","author":"Seide","year":"2014","journal-title":"Interspeech 2014"},{"key":"10.1016\/j.future.2026.108498_bib0024","article-title":"QSGD: communication-efficient SGD via gradient quantization and encoding","volume":"30","author":"Alistarh","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.future.2026.108498_bib0025","series-title":"Proceedings of ICLR","article-title":"Decentralized deep learning with arbitrary communication compression","author":"Koloskova","year":"2020"},{"issue":"12","key":"10.1016\/j.future.2026.108498_bib0026","doi-asserted-by":"crossref","first-page":"3294","DOI":"10.1109\/TPDS.2023.3323282","article-title":"Communication optimization algorithms for distributed deep learning systems: a survey","volume":"34","author":"Yu","year":"2023","journal-title":"IEEE Trans. Parall. Distrib. Syst."},{"key":"10.1016\/j.future.2026.108498_bib0027","article-title":"Terngrad: ternary gradients to reduce communication in distributed deep learning","volume":"30","author":"Wen","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.future.2026.108498_bib0028","series-title":"Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining","first-page":"2135","article-title":"CNTK: Microsoft\u2019s open-source deep-learning toolkit","author":"Seide","year":"2016"},{"key":"10.1016\/j.future.2026.108498_bib0029","unstructured":"T. Chen, M. Li, Y. Li, M. Lin, N. Wang, M. Wang, T. Xiao, B. Xu, C. Zhang, Z. Zhang, Mxnet: a flexible and efficient machine learning library for heterogeneous distributed systems, (2015). arXiv: 1512.01274."},{"key":"10.1016\/j.future.2026.108498_bib0030","series-title":"Proceedings of the Eleventh European Conference on Computer Systems","first-page":"1","article-title":"Geeps: scalable deep learning on distributed gpus with a gpu-specialized parameter server","author":"Cui","year":"2016"},{"key":"10.1016\/j.future.2026.108498_bib0031","series-title":"Proceedings of the 17th European Conference on Computer Systems","first-page":"435","article-title":"Out-of-order backprop: an effective scheduling technique for deep learning","author":"Oh","year":"2022"},{"key":"10.1016\/j.future.2026.108498_bib0032","series-title":"Proceedings of the Computational Science - ICCS 2004","first-page":"1","article-title":"Optimization of collective reduction operations","author":"Rabenseifner","year":"2004"},{"key":"10.1016\/j.future.2026.108498_bib0033","series-title":"International Conference on Machine Learning","first-page":"8852","article-title":"DAdaQuant: doubly-adaptive quantization for communication-efficient federated learning","author":"H\u00f6nig","year":"2022"},{"key":"10.1016\/j.future.2026.108498_bib0034","series-title":"2021 IEEE 18th International Conference on Mobile Ad Hoc and Smart Systems (MASS)","first-page":"136","article-title":"DQ-SGD: Dynamic quantization in SGD for communication-efficient distributed learning","author":"Yan","year":"2021"},{"issue":"2","key":"10.1016\/j.future.2026.108498_bib0035","doi-asserted-by":"crossref","first-page":"1104","DOI":"10.1109\/JSYST.2024.3381590","article-title":"Efficient and flexible component placement for serverless computing","volume":"18","author":"Luo","year":"2024","journal-title":"IEEE Syst. J."},{"issue":"1","key":"10.1016\/j.future.2026.108498_bib0036","doi-asserted-by":"crossref","first-page":"178","DOI":"10.1109\/TNET.2022.3187821","article-title":"Meeting coflow deadlines in data center networks with policy-based selective completion","volume":"31","author":"Luo","year":"2023","journal-title":"IEEE\/ACM Trans. Network."},{"key":"10.1016\/j.future.2026.108498_bib0037","series-title":"Proceedings of the ACM SIGCOMM 2010 Conference, SIGCOMM \u201910","first-page":"63","article-title":"Data center TCP (DCTCP)","author":"Alizadeh","year":"2010"},{"issue":"6","key":"10.1016\/j.future.2026.108498_bib0038","doi-asserted-by":"crossref","first-page":"380","DOI":"10.1109\/MNET.2024.3397781","article-title":"RDMA Transports in datacenter networks: survey","volume":"38","author":"Hu","year":"2024","journal-title":"IEEE Netw."},{"key":"10.1016\/j.future.2026.108498_bib0039","first-page":"1","article-title":"IEEE Standard for floating-point arithmetic","year":"2019","journal-title":"IEEE Std 754-2019 (Revision of IEEE 754-2008)"},{"key":"10.1016\/j.future.2026.108498_bib0040","series-title":"Proceedings of the ACM SIGCOMM 2021 Conference","first-page":"657","article-title":"SiP-ML: High-bandwidth optical network interconnects for machine learning training","author":"Khani","year":"2021"},{"key":"10.1016\/j.future.2026.108498_bib0041","series-title":"2023 IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS)","first-page":"283","article-title":"ASTRA-sim2.0: modeling hierarchical networks and disaggregated systems for large-model training at scale","author":"Won","year":"2023"},{"key":"10.1016\/j.future.2026.108498_bib0042","series-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, SC \u201921","article-title":"Characterization and prediction of deep learning workloads in large-scale GPU datacenters","author":"Hu","year":"2021"},{"key":"10.1016\/j.future.2026.108498_bib0043","series-title":"Proceedings of the 21st USENIX Symposium on Networked Systems Design and Implementation, NSDI\u201924","article-title":"Characterization of large language model development in the datacenter","author":"Hu","year":"2024"},{"key":"10.1016\/j.future.2026.108498_bib0044","series-title":"2019 IEEE International Symposium on Workload Characterization (IISWC)","first-page":"189","article-title":"Characterizing deep learning training workloads on Alibaba-PAI","author":"Wang","year":"2019"},{"issue":"6","key":"10.1016\/j.future.2026.108498_bib0045","doi-asserted-by":"crossref","first-page":"1:1","DOI":"10.1147\/JRD.2019.2947013","article-title":"BlueConnect: decomposing all-reduce for deep learning on heterogeneous network hierarchy","volume":"63","author":"Cho","year":"2019","journal-title":"IBM J. Res. Dev."},{"key":"10.1016\/j.future.2026.108498_bib0046","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.future.2026.108498_bib0047","series-title":"Proceedings of the 25th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","first-page":"45","article-title":"Taming unbalanced training workloads in deep learning with partial collective operations","author":"Li","year":"2020"},{"key":"10.1016\/j.future.2026.108498_bib0048","unstructured":"A. Samajdar, Y. Zhu, P. Whatmough, M. Mattina, T. Krishna, Scale-sim: Systolic cnn accelerator simulator, (2018). arXiv: 1811.02883."},{"key":"10.1016\/j.future.2026.108498_bib0049","series-title":"Proceedings of the 2024 IEEE\/ACM 32nd International Symposium on Quality of Service (IWQoS)","first-page":"1","article-title":"Towards optimal topology-aware AllReduce synthesis","author":"Lv","year":"2024"}],"container-title":["Future Generation Computer Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167739X26001329?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167739X26001329?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T01:38:01Z","timestamp":1780018681000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167739X26001329"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":49,"alternative-id":["S0167739X26001329"],"URL":"https:\/\/doi.org\/10.1016\/j.future.2026.108498","relation":{},"ISSN":["0167-739X"],"issn-type":[{"value":"0167-739X","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Maximizing the computation-communication overlap for distributed deep learning with approximate AllReduce","name":"articletitle","label":"Article Title"},{"value":"Future Generation Computer Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.future.2026.108498","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"108498"}}