{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T10:17:27Z","timestamp":1784024247884,"version":"3.55.0"},"reference-count":46,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2021,7,14]],"date-time":"2021-07-14T00:00:00Z","timestamp":1626220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,7,14]],"date-time":"2021-07-14T00:00:00Z","timestamp":1626220800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100012165","name":"Key Technologies Research and Development Program","doi-asserted-by":"publisher","award":["2018AAA0103203"],"award-info":[{"award-number":["2018AAA0103203"]}],"id":[{"id":"10.13039\/501100012165","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2022,3]]},"DOI":"10.1007\/s10489-021-02588-9","type":"journal-article","created":{"date-parts":[[2021,7,14]],"date-time":"2021-07-14T04:03:22Z","timestamp":1626235402000},"page":"3880-3900","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["Optimal distributed parallel algorithms for deep learning framework Tensorflow"],"prefix":"10.1007","volume":"52","author":[{"given":"Yuanlun","family":"Xie","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Majun","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tingsong","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5551-9796","authenticated-orcid":false,"given":"Wenhong","family":"Tian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,7,14]]},"reference":[{"issue":"7553","key":"2588_CR1","doi-asserted-by":"publisher","first-page":"436","DOI":"10.1038\/nature14539","volume":"521","author":"Y LeCun","year":"2015","unstructured":"LeCun Y, Bengio Y, Hinton G (2015) Deep learning. Nature 521(7553):436\u2013444","journal-title":"Nature"},{"issue":"5786","key":"2588_CR2","doi-asserted-by":"publisher","first-page":"504","DOI":"10.1126\/science.1127647","volume":"313","author":"GE Hinton","year":"2006","unstructured":"Hinton GE, Salakhutdinov RR (2006) Reducing the dimensionality of data with neural networks. Science 313(5786):504\u2013507","journal-title":"Science"},{"key":"2588_CR3","unstructured":"Brownlee J (2018) Better deep learning: train faster, reduce overfitting, and make better predictions machine learning mastery"},{"key":"2588_CR4","unstructured":"Shanmugamani R (2018) Deep learning for computer vision: expert techniques to train advanced neural networks using Tensorflow and Keras. Packt Publishing Ltd"},{"key":"2588_CR5","unstructured":"Hendrycks D, Mazeika M, Wilson D, Gimpel K (2018) Using trusted data to train deep networks on labels corrupted by severe noise. In: Advances in neural information processing systems, pp 10456\u201310465"},{"key":"2588_CR6","doi-asserted-by":"crossref","unstructured":"Miikkulainen R, Liang J, Meyerson E, Rawal A, Fink D, Francon O, Raju B, Shahrzad H, Navruzyan A, Duffy N et al (2019) Evolving deep neural networks. In: Artificial intelligence in the age of neural networks and brain computing. Elsevier, pp 293\u2013312","DOI":"10.1016\/B978-0-12-815480-9.00015-3"},{"key":"2588_CR7","doi-asserted-by":"publisher","first-page":"257","DOI":"10.1016\/j.ecoinf.2018.10.002","volume":"48","author":"BB Traore","year":"2018","unstructured":"Traore BB, Kamsu-Foguem B, Tangara F (2018) Deep convolution neural network for image recognition. Ecological Informatics 48:257\u2013268","journal-title":"Ecological Informatics"},{"issue":"1","key":"2588_CR8","first-page":"1","volume":"8","author":"Z Che","year":"2018","unstructured":"Che Z, Purushotham S, Cho K, Sontag D, Liu Y (2018) Recurrent neural networks for multivariate time series with missing values. Scientific Reports 8(1):1\u201312","journal-title":"Scientific Reports"},{"key":"2588_CR9","unstructured":"Gu J, Chowdhury M, Shin KG, Zhu Y, Jeon M, Qian J, Liu H, Guo C (2019) Tiresias: a {GPU} cluster manager for distributed deep learning. In: 16th {USENIX} symposium on networked systems design and implementation ({NSDI}, vol 19, pp 485\u2013500"},{"key":"2588_CR10","doi-asserted-by":"crossref","unstructured":"Shi S, Wang Q, Chu X, Li B, Qin Y, Liu R, Zhao X (2020) Communication-efficient distributed deep learning with merged gradient sparsification on gpus. In: IEEE INFOCOM","DOI":"10.1109\/INFOCOM41043.2020.9155269"},{"key":"2588_CR11","doi-asserted-by":"crossref","unstructured":"Malik A, Lu M, Wang N, Lin Y, Yoo S (2018) Detailed performance analysis of distributed Tensorflow on a gpu cluster using deep learning algorithms. In: 2018 New York scientific data summit (NYSDS). IEEE, pp 1\u20138","DOI":"10.1109\/NYSDS.2018.8538946"},{"key":"2588_CR12","doi-asserted-by":"crossref","unstructured":"Chen C, Weng Q, Wang W, Li B, Li B (2018) Fast distributed deep learning via worker-adaptive batch sizing. In: Proceedings of the ACM symposium on cloud computing, pp 521\u2013521","DOI":"10.1145\/3267809.3275463"},{"key":"2588_CR13","doi-asserted-by":"crossref","unstructured":"Yang E, Kim S-H, Kim T-W, Jeon M, Park S, Youn C-H (2018) An adaptive batch-orchestration algorithm for the heterogeneous gpu cluster environment in distributed deep learning system. In: 2018 IEEE international conference on big data and smart computing (BigComp). IEEE, pp 725\u2013728","DOI":"10.1109\/BigComp.2018.00136"},{"key":"2588_CR14","doi-asserted-by":"crossref","unstructured":"Bao Y, Peng Y, Wu C (2019) Deep learning-based job placement in distributed machine learning clusters. In: IEEE INFOCOM 2019-IEEE conference on computer communications. IEEE, pp 505\u2013513","DOI":"10.1109\/INFOCOM.2019.8737460"},{"issue":"2","key":"2588_CR15","doi-asserted-by":"publisher","first-page":"227","DOI":"10.3102\/1076998619872761","volume":"45","author":"B Pang","year":"2020","unstructured":"Pang B, Nijkamp E, Wu YN (2020) Deep learning with Tensorflow: a review. J Educ Behav Stat 45(2):227\u2013248","journal-title":"J Educ Behav Stat"},{"key":"2588_CR16","doi-asserted-by":"crossref","unstructured":"Seetala K, Birdsong W, Reddy YB (2019) Image classification using Tensorflow. In: 16th international conference on information technology-new generations (ITNG 2019). Springer, pp 485\u2013488","DOI":"10.1007\/978-3-030-14070-0_67"},{"key":"2588_CR17","unstructured":"Dean J, Corrado G, Monga R, Chen K, Devin M, Mao M, Ranzato M, Senior A, Tucker P, Yang K et al (2012) Large scale distributed deep networks. In: Advances in neural information processing systems, pp 1223\u20131231"},{"key":"2588_CR18","doi-asserted-by":"publisher","first-page":"78","DOI":"10.1016\/j.artint.2014.02.004","volume":"210","author":"P Baldi","year":"2014","unstructured":"Baldi P, Sadowski P (2014) The dropout learning algorithm. Artificial Intelligence 210:78\u2013122","journal-title":"Artificial Intelligence"},{"issue":"1","key":"2588_CR19","doi-asserted-by":"publisher","first-page":"16","DOI":"10.1186\/s40537-019-0179-2","volume":"6","author":"RK Kennedy","year":"2019","unstructured":"Kennedy RK, Khoshgoftaar TM, Villanustre F, Humphrey T (2019) A parallel and distributed stochastic gradient descent implementation using commodity clusters. Journal of Big Data 6(1):16","journal-title":"Journal of Big Data"},{"key":"2588_CR20","doi-asserted-by":"crossref","unstructured":"Du X, Kuang D, Ye Y, Li X, Chen M, Du Y, Wu W (2018) Comparative study of distributed deep learning tools on supercomputers. In: International conference on algorithms and architectures for parallel processing. Springer, pp 122\u2013137","DOI":"10.1007\/978-3-030-05051-1_9"},{"key":"2588_CR21","doi-asserted-by":"crossref","unstructured":"Kang B, Jeong J-H, Jeong C (2018) Distributed parallel deep learning for fast extraction of similar weather map. In: TENCON 2018-2018 IEEE region 10 conference. IEEE, pp 1426\u20131429","DOI":"10.1109\/TENCON.2018.8650104"},{"key":"2588_CR22","doi-asserted-by":"crossref","unstructured":"Li D, Lai Z, Ge K, Zhang Y, Zhang Z, Wang Q, Wang H (2019) Hpdl: towards a general framework for high-performance distributed deep learning. In: 2019 IEEE 39th International Conference on Distributed Computing Systems (ICDCS). IEEE, pp 1742\u20131753","DOI":"10.1109\/ICDCS.2019.00173"},{"key":"2588_CR23","doi-asserted-by":"crossref","unstructured":"Kim S, Yu G-I, Park H, Cho S, Jeong E, Ha H, Lee S, Jeong JS, Chun B-G (2019) Parallax: sparsity-aware data parallel training of deep neural networks. In: Proceedings of the fourteenth eurosys conference 2019, pp 1\u201315","DOI":"10.1145\/3302424.3303957"},{"issue":"04","key":"2588_CR24","doi-asserted-by":"publisher","first-page":"1950022","DOI":"10.1142\/S1469026819500226","volume":"18","author":"DJ Gunn","year":"2019","unstructured":"Gunn DJ, Liu Z, Dave R, Yuan X, Roy K (2019) Touch-based active dloud authentication using traditional machine learning and LSTM on a distributed Tensorflow framework. International Journal of Computational Intelligence and Applications 18(04):1950022","journal-title":"International Journal of Computational Intelligence and Applications"},{"key":"2588_CR25","doi-asserted-by":"crossref","unstructured":"Ranbirsingh JK, Kimm H, Kimm H (2019) Distributed neural networks using Tensorflow over multicore and many-core systems. In: 2019 IEEE 13th international symposium on embedded multicore\/many-core systems-on-chip (MCSoC). IEEE, pp 101\u2013107","DOI":"10.1109\/MCSoC.2019.00022"},{"issue":"32","key":"2588_CR26","first-page":"256","volume":"4","author":"RKL Kennedy","year":"2018","unstructured":"Kennedy RKL (2018) Parallel distributed deep learning on cluster computers. Training 4(32):256","journal-title":"Training"},{"issue":"9-10","key":"2588_CR27","doi-asserted-by":"publisher","first-page":"822","DOI":"10.1080\/08839514.2018.1508814","volume":"32","author":"J Marques","year":"2018","unstructured":"Marques J, Falcao G, Alexandre LA (2018) Distributed learning of cnns on heterogeneous cpu\/gpu architectures. Appl Artif Intell 32(9-10):822\u2013844","journal-title":"Appl Artif Intell"},{"key":"2588_CR28","doi-asserted-by":"crossref","unstructured":"Grabaskas N (2019) Improving usability of distributed neural network training. In: Intelligent computing-proceedings of the computing conference. Springer, pp 867\u2013886","DOI":"10.1007\/978-3-030-22871-2_62"},{"key":"2588_CR29","unstructured":"Wen W, Xu C, Yan F, Wu C, Wang Y, Chen Y, Li H (2017) Terngrad: ternary gradients to reduce communication in distributed deep learning. In: Advances in neural information processing systems, pp 1509\u20131519"},{"issue":"4","key":"2588_CR30","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3320060","volume":"52","author":"T Ben-Nun","year":"2019","unstructured":"Ben-Nun T, Hoefler T (2019) Demystifying parallel and distributed deep learning: an in-depth concurrency analysis. ACM Computing Surveys (CSUR) 52(4):1\u201343","journal-title":"ACM Computing Surveys (CSUR)"},{"issue":"8","key":"2588_CR31","doi-asserted-by":"publisher","first-page":"945","DOI":"10.1093\/jamia\/ocy017","volume":"25","author":"K Chang","year":"2018","unstructured":"Chang K, Balachandar N, Lam C, Yi D, Brown J, Beers A, Rosen B, Rubin DL, Kalpathy-Cramer J (2018) Distributed deep learning networks among institutions for medical imaging. J Am Med Inform Assoc 25(8):945\u2013954","journal-title":"J Am Med Inform Assoc"},{"key":"2588_CR32","unstructured":"Chen C, Yang C, Cheng H (2018) Efficient and robust parallel dnn training through model parallelism on multi-gpu platform, arxiv: Distributed, Parallel and Cluster Computing"},{"key":"2588_CR33","doi-asserted-by":"publisher","unstructured":"Peng Y, Zhu Y, Chen Y, Bao Y, Yi B, Lan C, Wu C, Guo C (2019) A generic communication scheduler for distributed dnn training acceleration. In: Proceedings of the 27th ACM symposium on operating systems principles, ser. SOSP \u201919. New York, NY, USA: Association for Computing Machinery, pp 16\u201329. [Online]. Available: https:\/\/doi.org\/10.1145\/3341301.3359642","DOI":"10.1145\/3341301.3359642"},{"key":"2588_CR34","doi-asserted-by":"crossref","unstructured":"Surya RY, Imam Kistijantoro A (2019) Dynamic resource allocation for distributed Tensorflow training in kubernetes cluster. In: 2019 international conference on data and software engineering (ICoDSE), pp 1\u20136","DOI":"10.1109\/ICoDSE48700.2019.9092758"},{"key":"2588_CR35","doi-asserted-by":"crossref","unstructured":"Mayer R, Mayer C, Laich L (2017) The Tensorflow partitioning and scheduling problem: it\u2019s the critical path! arxiv: Distributed, Parallel, and Cluster Computing, pp 1\u20136","DOI":"10.1145\/3154842.3154843"},{"key":"2588_CR36","doi-asserted-by":"publisher","unstructured":"Chen C, Weng Q, Wang W, Li B, Li B (2018) Fast distributed deep learning via worker-adaptive batch sizing. In: Proceedings of the ACM symposium on cloud computing, ser. SoCC \u201918. New York, NY, USA: Association for Computing Machinery, p 521. [Online]. Available: https:\/\/doi.org\/10.1145\/3267809.3275463","DOI":"10.1145\/3267809.3275463"},{"key":"2588_CR37","doi-asserted-by":"publisher","unstructured":"Liu J, Jia C, Chen J, Lin H, Jin X, An H (2019) An effective method for operations placement in tensor flow. In: Proceedings of the 3rd international conference on high performance compilation, computing and communications, ser. HP3C \u201919. New York, NY, USA: Association for Computing Machinery, pp 13\u201319. [Online]. Available: https:\/\/doi.org\/10.1145\/3318265.3318270","DOI":"10.1145\/3318265.3318270"},{"key":"2588_CR38","unstructured":"Sergeev A, Del Balso M (2018) Horovod: fast and easy distributed deep learning in Tensorflow. arXiv:1802.05799"},{"issue":"2","key":"2588_CR39","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3296957.3173171","volume":"53","author":"D Fujiki","year":"2018","unstructured":"Fujiki D, Mahlke S, Das R (2018) In-memory data parallel processor. ACM SIGPLAN Not 53(2):1\u201314","journal-title":"ACM SIGPLAN Not"},{"key":"2588_CR40","doi-asserted-by":"crossref","unstructured":"Bienia C, Kumar S, Singh JP, Li K (2008) The parsec benchmark suite: characterization and architectural implications. In: Proceedings of the 17th international conference on parallel architectures and compilation techniques, pp 72\u201381","DOI":"10.1145\/1454115.1454128"},{"issue":"s2","key":"2588_CR41","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1515\/pomr-2017-0062","volume":"24","author":"Z Hu","year":"2017","unstructured":"Hu Z, Qin W (2017) Fuzzy method and neural network model parallel implementation of multi-layer neural network based on cloud computing for real time data transmission in large offshore platform. Polish Maritime Research 24(s2):39\u201344","journal-title":"Polish Maritime Research"},{"issue":"16","key":"2588_CR42","doi-asserted-by":"publisher","first-page":"e4989","DOI":"10.1002\/cpe.4989","volume":"31","author":"T Kurth","year":"2019","unstructured":"Kurth T, Smorkalov M, Mendygral P, Sridharan S, Mathuriya A (2019) Tensorflow at scale: performance and productivity analysis of distributed training with horovod, mlsl, and cray pe ml. Concurrency and Computation: Practice and Experience 31(16):e4989","journal-title":"Concurrency and Computation: Practice and Experience"},{"key":"2588_CR43","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1016\/j.cageo.2018.12.007","volume":"124","author":"M Liu","year":"2019","unstructured":"Liu M, Grana D (2019) Accelerating geostatistical seismic inversion using Tensorflow: a heterogeneous distributed deep learning framework. Computers & Geosciences 124:37\u201345","journal-title":"Computers & Geosciences"},{"key":"2588_CR44","doi-asserted-by":"crossref","unstructured":"Li M, Andersen DG, Park JW, Smola AJ, Ahmed A, Josifovski V, Long J, Shekita EJ, Su B-Y (2014) Scaling distributed machine learning with the parameter server. In: 11th {USENIX} symposium on operating systems design and implementation ({OSDI}, vol 14, pp 583\u2013598","DOI":"10.1145\/2640087.2644155"},{"key":"2588_CR45","unstructured":"Gibiansky A (2017) Bringing hpc techniques to deep learning, Baidu Research, Tech. Rep."},{"key":"2588_CR46","unstructured":"Abadi M, Agarwal A, Barham P, Brevdo E, Chen Z, Citro C, Corrado G, Davis A, Dean J, Devin M, Ghemawat S, Goodfellow I, Harp A, Irving G, Isard M, Jia Y, Kaiser L, Kudlur M, Levenberg J, Zheng X (2015) Tensorflow : large-scale machine learning on heterogeneous distributed systems, 01"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-021-02588-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-021-02588-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-021-02588-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,2,22]],"date-time":"2022-02-22T06:45:33Z","timestamp":1645512333000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-021-02588-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,7,14]]},"references-count":46,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2022,3]]}},"alternative-id":["2588"],"URL":"https:\/\/doi.org\/10.1007\/s10489-021-02588-9","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,7,14]]},"assertion":[{"value":"2 June 2021","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 July 2021","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Author Yuanlun Xie declares that he has no conflict of interest. Author Majun He declares that he has no conflict of interest. Author Tingsong Ma declares that he has no conflict of interest. Wenhong Tian declares that he has no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Conflict of Interests"}}]}}