{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T18:04:51Z","timestamp":1780423491820,"version":"3.54.1"},"reference-count":74,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Parallel Computing"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.parco.2026.103196","type":"journal-article","created":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T23:15:42Z","timestamp":1776467742000},"page":"103196","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["A survey of parallel computing frameworks and optimizations for AI and deep learning"],"prefix":"10.1016","volume":"128","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7803-0179","authenticated-orcid":false,"given":"Mohamed Nazih","family":"Omri","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.parco.2026.103196_b1","series-title":"AI and compute","author":"Amodei","year":"2018"},{"key":"10.1016\/j.parco.2026.103196_b2","series-title":"GPT-4 Technical Report","author":"OpenAI","year":"2023"},{"key":"10.1016\/j.parco.2026.103196_b3","series-title":"PaLM 2 technical report","author":"Anil","year":"2023"},{"issue":"2196","key":"10.1016\/j.parco.2026.103196_b4","article-title":"The future of computing beyond Moore\u2019s law","volume":"379","author":"Shalf","year":"2021","journal-title":"Phil. Trans. R. Soc. A"},{"key":"10.1016\/j.parco.2026.103196_b5","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019","journal-title":"NAACL"},{"key":"10.1016\/j.parco.2026.103196_b6","first-page":"1","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","volume":"24","author":"Fedus","year":"2021","journal-title":"JMLR"},{"issue":"3","key":"10.1016\/j.parco.2026.103196_b7","first-page":"1","article-title":"Ten lessons from three generations shaped google\u2019s TPUv4i","volume":"49","author":"Jouppi","year":"2021","journal-title":"ACM SIGARCH Comput. Archit. News"},{"key":"10.1016\/j.parco.2026.103196_b8","series-title":"NVIDIA Grace Hopper Superchip Architecture","author":"NVIDIA Corporation","year":"2023"},{"issue":"4","key":"10.1016\/j.parco.2026.103196_b9","doi-asserted-by":"crossref","first-page":"8","DOI":"10.1109\/MM.2019.2920814","article-title":"DeepSpeed: Extreme-scale model training for everyone","volume":"43","author":"Yao","year":"2023","journal-title":"IEEE Micro"},{"key":"10.1016\/j.parco.2026.103196_b10","series-title":"Megatron-LM: Training multi-billion parameter language models using model parallelism","author":"Shoeybi","year":"2020"},{"key":"10.1016\/j.parco.2026.103196_b11","doi-asserted-by":"crossref","unstructured":"D. Narayanan, M. Shoeybi, J. Casper, P. LeGresley, M. Patwary, V. Korthikanti, D. Vainbrand, P. Kashinkunti, J. Bernauer, B. Catanzaro, et al., Efficient large-scale language model training on gpu clusters, in: Proceedings of Machine Learning and Systems, vol. 4, 2021, pp. 1\u201315.","DOI":"10.1145\/3458817.3476209"},{"key":"10.1016\/j.parco.2026.103196_b12","first-page":"1","article-title":"A survey on efficient training of transformers","volume":"24","author":"Wang","year":"2023","journal-title":"JMLR"},{"key":"10.1016\/j.parco.2026.103196_b13","first-page":"367","article-title":"Deep learning with coherent nanophotonic circuits","volume":"15","author":"Shen","year":"2021","journal-title":"Nat. Photonics"},{"issue":"7","key":"10.1016\/j.parco.2026.103196_b14","first-page":"18","article-title":"The carbon footprint of machine learning training Will Plateau, then Shrink","volume":"55","author":"Patterson","year":"2022","journal-title":"Comput."},{"issue":"5","key":"10.1016\/j.parco.2026.103196_b15","first-page":"1","article-title":"A survey of distributed deep learning systems","volume":"53","author":"Smith","year":"2020","journal-title":"ACM Comput. Surv."},{"issue":"150","key":"10.1016\/j.parco.2026.103196_b16","first-page":"1","article-title":"Scaling deep learning: Models, systems, and challenges","volume":"22","author":"Johnson","year":"2021","journal-title":"J. Mach. Learn. Res."},{"issue":"4","key":"10.1016\/j.parco.2026.103196_b17","first-page":"801","article-title":"Memory and communication optimizations for parallel deep learning","volume":"33","author":"Chen","year":"2022","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"10.1016\/j.parco.2026.103196_b18","series-title":"Advances in Neural Information Processing Systems","first-page":"5998","article-title":"Attention is all you need","volume":"vol. 30","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.parco.2026.103196_b19","series-title":"B200 Blackwell Architecture and Performance","author":"NVIDIA Corporation","year":"2025"},{"key":"10.1016\/j.parco.2026.103196_b20","series-title":"TPU v5: Scaling performance and efficiency for mixture-of-experts models","author":"Google Research","year":"2025"},{"key":"10.1016\/j.parco.2026.103196_b21","doi-asserted-by":"crossref","unstructured":"L. Chen, A. Patel, M. Schmidt, The great consolidation: A 2025 retrospective on deep learning framework ecosystems, in: Proceedings of the MLSys Conference, vol. 4, 2025, pp. 1\u201315.","DOI":"10.1063\/5.0290691"},{"key":"10.1016\/j.parco.2026.103196_b22","unstructured":"M. Li, D.G. Andersen, J.W. Park, A.J. Smola, A. Ahmed, V. Josifovski, J. Long, E.J. Shekita, B.-Y. Su, Scaling distributed machine learning with the parameter server, in: Proceedings of the USENIX Symposium on Operating Systems Design and Implementation, OSDI, vol. 14, 2014, pp. 583\u2013598."},{"key":"10.1016\/j.parco.2026.103196_b23","series-title":"SC20: International Conference for High Performance Computing, Networking, Storage and Analysis","first-page":"1","article-title":"Zero: Memory optimizations toward training trillion parameter models","author":"Rajbhandari","year":"2020"},{"issue":"2","key":"10.1016\/j.parco.2026.103196_b24","doi-asserted-by":"crossref","first-page":"117","DOI":"10.1016\/j.jpdc.2008.09.002","article-title":"Bandwidth optimal all-reduce algorithms for clusters of workstations","volume":"69","author":"Patarasuk","year":"2009","journal-title":"J. Parallel Distrib. Comput."},{"key":"10.1016\/j.parco.2026.103196_b25","unstructured":"J. Ren, S. Rajbhandari, J. Rasley, O. Ruwase, Y. He, Zero-offload: Democratizing billion-scale model training, in: 2021 USENIX Annual Technical Conference, USENIX ATC 21, 2021, pp. 551\u2013564."},{"key":"10.1016\/j.parco.2026.103196_b26","unstructured":"M. Li, D.G. Andersen, J.W. Park, A.J. Smola, A. Ahmed, V. Josifovski, J. Long, E.J. Shekita, B.-Y. Su, Scaling distributed machine learning with the parameter server, in: Proceedings of the 11th USENIX Symposium on Operating Systems Design and Implementation, OSDI 14, 2014, pp. 583\u2013598."},{"key":"10.1016\/j.parco.2026.103196_b27","series-title":"Artificial Intelligence and Statistics","first-page":"1273","article-title":"Communication-efficient learning of deep networks from decentralized data","author":"McMahan","year":"2017"},{"key":"10.1016\/j.parco.2026.103196_b28","first-page":"8024","article-title":"Pytorch: An imperative style, high-performance deep learning library","volume":"32","author":"Paszke","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.parco.2026.103196_b29","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.parco.2026.103196_b30","series-title":"Zero-infinity: Breaking the gpu memory wall for extreme scale deep learning","author":"Rajbhandari","year":"2021"},{"key":"10.1016\/j.parco.2026.103196_b31","unstructured":"T. Chen, T. Moreau, Z. Jiang, L. Zheng, E. Yan, H. Shen, M. Cowan, L. Wang, Y. Hu, L. Ceze, et al., TVM: An automated End-to-End optimizing compiler for deep learning, in: 13th USENIX Symposium on Operating Systems Design and Implementation, OSDI 18, 2018, pp. 578\u2013594."},{"key":"10.1016\/j.parco.2026.103196_b32","series-title":"NVIDIA cuDNN: The GPU-Accelerated Deep Learning Library","author":"NVIDIA Corporation","year":"2023"},{"key":"10.1016\/j.parco.2026.103196_b33","doi-asserted-by":"crossref","first-page":"16344","DOI":"10.52202\/068431-1189","article-title":"Flashattention: Fast and memory-efficient exact attention with io-awareness","volume":"35","author":"Dao","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst. (NeurIPS)"},{"key":"10.1016\/j.parco.2026.103196_b34","series-title":"Mixed precision training","author":"Micikevicius","year":"2018"},{"key":"10.1016\/j.parco.2026.103196_b35","unstructured":"M. Abadi, P. Barham, J. Chen, Z. Chen, A. Davis, J. Dean, M. Devin, S. Ghemawat, G. Irving, M. Isard, et al., Tensorflow: A system for large-scale machine learning, in: 12th USENIX Symposium on Operating Systems Design and Implementation, OSDI 16, 2016, pp. 265\u2013283."},{"issue":"4","key":"10.1016\/j.parco.2026.103196_b36","first-page":"1115","article-title":"Full-stack optimization for accelerator-agnostic deep learning","volume":"34","author":"Li","year":"2023","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"10.1016\/j.parco.2026.103196_b37","unstructured":"A. Sharma, W. Liu, S. Rajbhandari, ZeRO+++: Dynamic Tensor Offloading with Predictive Prefetching for Trillion-Parameter Training, in: Proceedings of the USENIX Symposium on Operating Systems Design and Implementation, OSDI, 2025, pp. 1\u201318."},{"key":"10.1016\/j.parco.2026.103196_b38","series-title":"Dynamo: Dynamic optimization for deep learning","author":"Huang","year":"2022"},{"key":"10.1016\/j.parco.2026.103196_b39","unstructured":"B. Lepers, G. Quenot, O. Aumage, S. Thibault, Capuchin: Tensor-based gpu memory management for deep learning, in: Proceedings of the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, vol. 3, 2023, pp. 1\u201315."},{"key":"10.1016\/j.parco.2026.103196_b40","unstructured":"A. Agrawal, A.N. Modi, A. Passos, A. Lavoie, A. Agarwal, A. Shankar, I. Ganichev, J. Levenberg, M. Hong, R. Monga, et al., Tensorflow eager: A multi-stage, python-embedded dsl for machine learning, in: Proceedings of Machine Learning and Systems, vol. 1, 2019, pp. 178\u2013189."},{"key":"10.1016\/j.parco.2026.103196_b41","series-title":"JAX: composable transformations of Python+NumPy programs","author":"Bradbury","year":"2018"},{"key":"10.1016\/j.parco.2026.103196_b42","series-title":"Adam: A method for stochastic optimization","author":"Kingma","year":"2014"},{"key":"10.1016\/j.parco.2026.103196_b43","series-title":"Training deep nets with sublinear memory cost","author":"Chen","year":"2016"},{"key":"10.1016\/j.parco.2026.103196_b44","series-title":"LLM.int8(): 8-bit matrix multiplication for transformers at scale","author":"Dettmers","year":"2022"},{"key":"10.1016\/j.parco.2026.103196_b45","series-title":"Adafactor: Adaptive learning rates with sublinear memory cost","author":"Shazeer","year":"2020"},{"key":"10.1016\/j.parco.2026.103196_b46","first-page":"9724","article-title":"Memory efficient adaptive optimization","volume":"33","author":"Anil","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.parco.2026.103196_b47","series-title":"Deep gradient compression: Reducing the communication bandwidth for distributed training","author":"Lin","year":"2017"},{"key":"10.1016\/j.parco.2026.103196_b48","series-title":"Advances in Neural Information Processing Systems","article-title":"Lossless 1-bit LION: Communication-efficient training without accuracy drop","volume":"vol. 38","author":"Lee","year":"2025"},{"key":"10.1016\/j.parco.2026.103196_b49","article-title":"More effective distributed ml via a stale synchronous parallel parameter server","volume":"26","author":"Ho","year":"2013","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.parco.2026.103196_b50","article-title":"Can decentralized algorithms outperform centralized algorithms? a case study for decentralized parallel stochastic gradient descent","volume":"30","author":"Lian","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.parco.2026.103196_b51","article-title":"Terngrad: Ternary gradients to reduce communication in distributed deep learning","volume":"30","author":"Wen","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.parco.2026.103196_b52","series-title":"Interspeech 2014","first-page":"1058","article-title":"1-bit stochastic gradient descent and its application to data-parallel distributed training of speech DNNs","author":"Seide","year":"2014"},{"key":"10.1016\/j.parco.2026.103196_b53","unstructured":"P. Jain, A. Zhang, K. Ajayi, F. Lai, A. Gholami, K. Keutzer, J. Gonzalez, I. Stoica, Criticality of communication in memory-efficient distributed deep learning, in: Proceedings of Machine Learning and Systems, vol. 3, 2021, pp. 1\u201316."},{"key":"10.1016\/j.parco.2026.103196_b54","series-title":"NVIDIA Collective Communications Library (NCCL)","author":"NVIDIA Corporation","year":"2023"},{"issue":"1","key":"10.1016\/j.parco.2026.103196_b55","doi-asserted-by":"crossref","first-page":"49","DOI":"10.1177\/1094342005051521","article-title":"Optimization of collective communication operations in mpich","volume":"19","author":"Thakur","year":"2005","journal-title":"Int. J. High Perform. Comput. Appl."},{"key":"10.1016\/j.parco.2026.103196_b56","series-title":"TensorFlow Lite: ML for Mobile and Edge Devices","author":"Google","year":"2023"},{"key":"10.1016\/j.parco.2026.103196_b57","series-title":"ExecuTorch: PyTorch for On-Device Inference","author":"Meta","year":"2023"},{"key":"10.1016\/j.parco.2026.103196_b58","series-title":"ONNX Runtime: Cross-Platform Inference Accelerator","author":"Microsoft","year":"2023"},{"key":"10.1016\/j.parco.2026.103196_b59","series-title":"TensorRT: High-Performance Deep Learning Inference","author":"NVIDIA Corporation","year":"2023"},{"key":"10.1016\/j.parco.2026.103196_b60","article-title":"Gpipe: Efficient training of giant neural networks using pipeline parallelism","volume":"32","author":"Huang","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.parco.2026.103196_b61","series-title":"Efficient large-scale language model training on gpu clusters","author":"Narang","year":"2021"},{"issue":"3","key":"10.1016\/j.parco.2026.103196_b62","first-page":"321","article-title":"Memory optimization in deep learning: A framework survey","volume":"111","author":"Brandt","year":"2023","journal-title":"Proc. IEEE"},{"key":"10.1016\/j.parco.2026.103196_b63","series-title":"Mxnet: A flexible and efficient machine learning library for heterogeneous distributed systems","author":"Chen","year":"2015"},{"key":"10.1016\/j.parco.2026.103196_b64","unstructured":"A. Sharma, W. Liu, S. Rajbhandari, ZeRO+++: dynamic tensor offloading with predictive prefetching for trillion-parameter training, in: Proceedings of the 19th USENIX Symposium on Operating Systems Design and Implementation, OSDI 25, 2025, pp. 1\u201318."},{"key":"10.1016\/j.parco.2026.103196_b65","series-title":"Interspeech","first-page":"1058","article-title":"1-bit stochastic gradient descent and its application to data-parallel distributed training of speech DNNs","author":"Seide","year":"2014"},{"issue":"2","key":"10.1016\/j.parco.2026.103196_b66","first-page":"45","article-title":"The rise of the AI superchip: A taxonomy of 2025 architectures and their framework support","volume":"45","author":"Singh","year":"2025","journal-title":"IEEE Micro"},{"key":"10.1016\/j.parco.2026.103196_b67","series-title":"NVIDIA H100 Tensor Core GPU Architecture","author":"NVIDIA Corporation","year":"2023"},{"key":"10.1016\/j.parco.2026.103196_b68","series-title":"Apple Neural Engine: Core ML Integration","author":"Apple","year":"2023"},{"key":"10.1016\/j.parco.2026.103196_b69","first-page":"789","article-title":"Photonic tensor cores: End-to-end optical neural network training at scale","volume":"631","author":"Wang","year":"2025","journal-title":"Nature"},{"key":"10.1016\/j.parco.2026.103196_b70","series-title":"Federated learning: Collaborative machine learning without centralized training data","author":"Google","year":"2016"},{"key":"10.1016\/j.parco.2026.103196_b71","series-title":"Split learning for distributed deep learning","author":"Gupta","year":"2022"},{"issue":"7671","key":"10.1016\/j.parco.2026.103196_b72","doi-asserted-by":"crossref","first-page":"195","DOI":"10.1038\/nature23474","article-title":"Quantum machine learning","volume":"549","author":"Biamonte","year":"2017","journal-title":"Nature"},{"key":"10.1016\/j.parco.2026.103196_b73","unstructured":"D. Team, DeepInspect: A Unified, Causality-Aware Platform for Debugging Distributed Training, in: Proceedings of the ACM Symposium on Operating Systems Principles, SOSP, 2025, pp. 1\u201317."},{"key":"10.1016\/j.parco.2026.103196_b74","series-title":"The carbon-aware training (CAT) protocol: A framework for sustainable AI","author":"EU AI Research Initiative","year":"2025"}],"container-title":["Parallel Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167819126000141?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167819126000141?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T17:29:10Z","timestamp":1780421350000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167819126000141"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":74,"alternative-id":["S0167819126000141"],"URL":"https:\/\/doi.org\/10.1016\/j.parco.2026.103196","relation":{},"ISSN":["0167-8191"],"issn-type":[{"value":"0167-8191","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"A survey of parallel computing frameworks and optimizations for AI and deep learning","name":"articletitle","label":"Article Title"},{"value":"Parallel Computing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.parco.2026.103196","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"103196"}}