{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,7]],"date-time":"2026-02-07T10:35:41Z","timestamp":1770460541416,"version":"3.49.0"},"reference-count":28,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2022,2,18]],"date-time":"2022-02-18T00:00:00Z","timestamp":1645142400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,2,18]],"date-time":"2022-02-18T00:00:00Z","timestamp":1645142400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100007129","name":"Natural Science Foundation of Shandong Province","doi-asserted-by":"publisher","award":["NO. ZR 2019MD034"],"award-info":[{"award-number":["NO. ZR 2019MD034"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Education Reform Project of Shandong Province","award":["M2020266"],"award-info":[{"award-number":["M2020266"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2022,3]]},"DOI":"10.1007\/s11042-022-12292-6","type":"journal-article","created":{"date-parts":[[2022,2,19]],"date-time":"2022-02-19T03:12:26Z","timestamp":1645240346000},"page":"11587-11604","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Multi-core parallel architecture design and experiment for deep learning model training"],"prefix":"10.1007","volume":"81","author":[{"given":"Li","family":"Wanwu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6305-1548","authenticated-orcid":false,"given":"Liu","family":"Lin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhang","family":"Jixian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liu","family":"Shuai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiu","family":"Jiahao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,2,18]]},"reference":[{"key":"12292_CR1","doi-asserted-by":"crossref","unstructured":"Ahn S, Kim J, Lim E, et al (2018) ShmCaffe: a distributed deep learning platform with shared memory buffer for HPC architecture [C]\/\/2018 IEEE 38th international conference on distributed computing systems (ICDCS). IEEE, 1118\u20131128","DOI":"10.1109\/ICDCS.2018.00111"},{"key":"12292_CR2","first-page":"2017","volume":"11351","author":"T Akiba","year":"1710","unstructured":"Akiba T, Fukuda K, Suzuki S (1710) ChainerMN: scalable distributed deep learning framework [J]. ArXiv preprint arXiv 11351:2017","journal-title":"ArXiv preprint arXiv"},{"key":"12292_CR3","first-page":"2016","volume":"05507","author":"A Aytekin","year":"1610","unstructured":"Aytekin A, Feyzmahdavian HR, Johansson M (1610) Analysis and implementation of an asynchronous optimization algorithm for the parameter server [J]. ArXiv preprint arXiv 05507:2016","journal-title":"ArXiv preprint arXiv"},{"issue":"3","key":"12292_CR4","doi-asserted-by":"publisher","first-page":"527","DOI":"10.1007\/s11554-018-0804-x","volume":"17","author":"Y Braham","year":"2020","unstructured":"Braham Y, Elloumi Y, Akil M, Bedoui MH (2020) Parallel computation of watershed transform in weighted graphs on shared memory machines [J]. J Real-Time Image Proc 17(3):527\u2013542","journal-title":"J Real-Time Image Proc"},{"issue":"4","key":"12292_CR5","doi-asserted-by":"publisher","first-page":"1624","DOI":"10.1109\/TITS.2019.2910295","volume":"21","author":"Y Chen","year":"2019","unstructured":"Chen Y, Lv Y, Wang FY (2019) Traffic flow imputation using parallel data and generative adversarial networks [J]. IEEE Trans Intell Transp Syst 21(4):1624\u20131630","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"12292_CR6","unstructured":"Dean J, Corrado GS, Monga R et al (2013) Large scale distributed deep networks [J]. Adv Neural Inf Proces Syst 1\u20139"},{"issue":"3","key":"12292_CR7","doi-asserted-by":"publisher","first-page":"623","DOI":"10.1007\/s11554-019-00870-1","volume":"16","author":"Y Dengpan","year":"2019","unstructured":"Dengpan Y, Shunzhi J, Shiyu L, ChangRui L (2019) Faster and transferable deep learning Steganalysis on GPU [J]. J Real-Time Image Proc 16(3):623\u2013633","journal-title":"J Real-Time Image Proc"},{"key":"12292_CR8","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1016\/j.jpdc.2019.05.016","volume":"133","author":"J Fang","year":"2019","unstructured":"Fang J, Fu H, Yang G, Hsieh CJ (2019) RedSync: reducing synchronization bandwidth for distributed deep learning training system [J]. Journal of Parallel and Distributed Computing 133:30\u201339","journal-title":"Journal of Parallel and Distributed Computing"},{"key":"12292_CR9","doi-asserted-by":"crossref","unstructured":"Farkas A, Kert\u00e9sz G, Lovas R (2020) Parallel and distributed training of deep neural networks: a brief overview [C]\/\/2020 IEEE 24th international conference on intelligent engineering systems (INES). IEEE, 165\u2013170","DOI":"10.1109\/INES49302.2020.9147123"},{"key":"12292_CR10","first-page":"2017","volume":"02677","author":"P Goyal","year":"1706","unstructured":"Goyal P, Doll\u00e1r P, Girshick R et al (1706) Accurate, large Minibatch Sgd: training Imagenet in 1 hour [J]. ArXiv preprint arXiv 02677:2017","journal-title":"ArXiv preprint arXiv"},{"key":"12292_CR11","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, et al (2016) Deep residual learning for image recognition [C]\/\/ IEEE conference on Computer Vision & Pattern Recognition. IEEE Computer Society","DOI":"10.1109\/CVPR.2016.90"},{"key":"12292_CR12","first-page":"2018","volume":"05358","author":"Z Jia","year":"1807","unstructured":"Jia Z, Zaharia M, Aiken A (1807) Beyond data and model parallelism for deep neural networks [J]. ArXiv preprint arXiv 05358:2018","journal-title":"ArXiv preprint arXiv"},{"key":"12292_CR13","first-page":"2016","volume":"04581","author":"PH Jin","year":"1611","unstructured":"Jin PH, Qiaochu Y, Iandola F et al (1611) How to Scale Distributed Deep Learning? ArXiv preprint arXiv 04581:2016","journal-title":"ArXiv preprint arXiv"},{"key":"12292_CR14","doi-asserted-by":"crossref","unstructured":"Kang BS, Jeong JH, Jeong CS (2018) Distributed Parallel Deep Learning for Fast Extraction of Similar Weather Map [C]\/\/TENCON 2018-2018 IEEE region 10 conference. IEEE 1426\u20131429","DOI":"10.1109\/TENCON.2018.8650104"},{"issue":"3","key":"12292_CR15","doi-asserted-by":"publisher","first-page":"2287","DOI":"10.1007\/s10586-020-03144-9","volume":"23","author":"Y Kim","year":"2020","unstructured":"Kim Y, Choi H, Lee J, Kim JS, Jei H, Roh H (2020) Towards an optimized distributed deep learning framework for a heterogeneous multi-GPU cluster [J]. Clust Comput 23(3):2287\u20132300","journal-title":"Clust Comput"},{"issue":"6","key":"12292_CR16","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1145\/3065386","volume":"60","author":"A Krizhevsky","year":"2017","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2017) ImageNet classification with deep convolutional neural networks [J]. Commun ACM 60(6):84\u201390","journal-title":"Commun ACM"},{"key":"12292_CR17","doi-asserted-by":"crossref","unstructured":"Lopez F, Chow E, Tomov S, et al (2020) Asynchronous SGD for DNN training on shared-memory parallel architectures [C]\/\/2020 IEEE international parallel and distributed processing symposium workshops (IPDPSW). IEEE, 1\u20134","DOI":"10.1109\/IPDPSW50202.2020.00168"},{"issue":"2","key":"12292_CR18","doi-asserted-by":"publisher","first-page":"117","DOI":"10.1016\/j.jpdc.2008.09.002","volume":"69","author":"P Patarasuk","year":"2009","unstructured":"Patarasuk P, Yuan X (2009) Bandwidth optimal all-reduce algorithms for clusters of workstations [J]. J Parallel Distr Com 69(2):117\u2013124","journal-title":"J Parallel Distr Com"},{"key":"12292_CR19","doi-asserted-by":"crossref","unstructured":"Seide F, Fu H, Droppo J, et al (2014) 1-Bit Stochastic Gradient Descent and Its Application to Data-Parallel Distributed Training of Speech DNNs [C]\/\/Fifteenth Annual Conference of the International Speech Communication Association","DOI":"10.21437\/Interspeech.2014-274"},{"key":"12292_CR20","unstructured":"Sermanet P, Eigen D, Zhang X et al (2013) OverFeat: integrated recognition, localization and detection using convolutional networks [J]. Eprint Arxiv 1\u201316"},{"key":"12292_CR21","doi-asserted-by":"crossref","unstructured":"Shi G, Zhang J, Zhang C et al (2020) A distributed parallel training method of deep belief networks [J]. Soft Comput 1\u201312","DOI":"10.1007\/s00500-020-04754-6"},{"key":"12292_CR22","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition [J]. Computer ence 1\u201314"},{"key":"12292_CR23","doi-asserted-by":"crossref","unstructured":"Sun P, Wen Y, Han R et al (2019) GradientFlow: Optimizing Network Performance for Large-Scale Distributed DNN Training.\u00a0IEEE T BIG DATA\u00a0(99):1\u20131","DOI":"10.1109\/TBDATA.2019.2957478"},{"key":"12292_CR24","doi-asserted-by":"crossref","unstructured":"Szegedy C, Wei L, Yangqing J, et al (2015) Going Deeper with Convolutions [C]. 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 1\u20139","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"12292_CR25","doi-asserted-by":"crossref","unstructured":"Zeiler MD, Fergus R (2014) Visualizing and Understanding Convolutional Networks [C]\/\/ European Conference on Computer Vision. Springer, Cham","DOI":"10.1007\/978-3-319-10590-1_53"},{"key":"12292_CR26","doi-asserted-by":"crossref","unstructured":"Zhang J, Zhan J, Li J, Jin J, Qian L (2020) Optimizing Execution for Pipelined-Based Distributed Deep Learning in a Heterogeneously Networked GPU Cluster.\u00a0Concurr Comp-Pract E\u00a0e5923:1\u201319","DOI":"10.1002\/cpe.5923"},{"issue":"12","key":"12292_CR27","doi-asserted-by":"publisher","first-page":"7369","DOI":"10.1109\/TII.2020.2976053","volume":"16","author":"Y Zhang","year":"2020","unstructured":"Zhang Y, Zhou Y, Lu H, Fujita H (2020) Traffic network flow prediction using parallel training for deep convolutional neural networks on spark cloud [J]. IEEE T IND INFORM 16(12):7369\u20137380","journal-title":"IEEE T IND INFORM"},{"key":"12292_CR28","first-page":"2020","volume":"12575","author":"W Zhu","year":"2006","unstructured":"Zhu W, Zhao C, Li W et al (2006) LAMP: large deep nets with automated model parallelism for image segmentation [J]. ArXiv preprint arXiv 12575:2020","journal-title":"ArXiv preprint arXiv"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-022-12292-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-022-12292-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-022-12292-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,3,29]],"date-time":"2022-03-29T11:46:47Z","timestamp":1648554407000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-022-12292-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,2,18]]},"references-count":28,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2022,3]]}},"alternative-id":["12292"],"URL":"https:\/\/doi.org\/10.1007\/s11042-022-12292-6","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,2,18]]},"assertion":[{"value":"28 February 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 January 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 January 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 February 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}}]}}