{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T10:53:53Z","timestamp":1781175233255,"version":"3.54.1"},"reference-count":26,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2019,4,16]],"date-time":"2019-04-16T00:00:00Z","timestamp":1555372800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,4,16]],"date-time":"2019-04-16T00:00:00Z","timestamp":1555372800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100010193","name":"Korea Electric Power Corporation","doi-asserted-by":"publisher","award":["R18XA05"],"award-info":[{"award-number":["R18XA05"]}],"id":[{"id":"10.13039\/501100010193","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010418","name":"Institute for Information and communications Technology Promotion","doi-asserted-by":"publisher","award":["2016-0-00087"],"award-info":[{"award-number":["2016-0-00087"]}],"id":[{"id":"10.13039\/501100010418","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2020,1]]},"DOI":"10.1007\/s11227-019-02845-2","type":"journal-article","created":{"date-parts":[[2019,4,16]],"date-time":"2019-04-16T05:03:46Z","timestamp":1555391026000},"page":"47-67","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":18,"title":["BOA: batch orchestration algorithm for straggler mitigation of distributed DL training in heterogeneous GPU cluster"],"prefix":"10.1007","volume":"76","author":[{"given":"Eunju","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dong-Ki","family":"Kang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chan-Hyun","family":"Youn","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2019,4,16]]},"reference":[{"key":"2845_CR1","unstructured":"Abadi M, Barham P, Chen J, Chen Z, Davis A, Dean J, Devin M, Ghemawat S, Irving G, Isard M, Kudlur M, Levenberg J, Monga R, Moore S, Murray DG, Steiner B, Tucker P, Vasudevan V, Warden P, Wicke M, Yu Y, Zheng X (2016) Tensorflow: a system for large-scale machine learning. In: Proceedings of the 12th USENIX Conference on Operating Systems Design and Implementation, OSDI\u201916, Berkeley, CA, USA. USENIX Association, pp 265\u2013283"},{"key":"2845_CR2","unstructured":"Boyd S, Mattingley J (2007) Branch and bound methods. Notes for EE364b, Stanford University, pp 2006\u20132007"},{"key":"2845_CR3","unstructured":"Chen J, Pan X, Monga R, Bengio S, Jozefowicz R (2016) Revisiting distributed synchronous SGD. arXiv preprint \narXiv:1604.00981"},{"key":"2845_CR4","unstructured":"Chetlur S, Woolley C, Vandermersch P, Cohen J, Tran J, Catanzaro B, Shelhamer E (2014) cudnn: efficient primitives for deep learning. arXiv preprint \narXiv:1410.0759"},{"key":"2845_CR5","unstructured":"Coates A, Huval B, Wang T, Wu D, Catanzaro B, Andrew N (2013) Deep learning with COTS HPC systems. In: International Conference on Machine Learning, pp 1337\u20131345"},{"key":"2845_CR6","unstructured":"Dean J, Corrado G, Monga R, Chen K, Devin M, Mao M, Senior A, Tucker P, Yang K, Le QV et\u00a0al (2012) Large scale distributed deep networks. In: Advances in neural information processing systems, pp 1223\u20131231"},{"issue":"83","key":"2845_CR7","first-page":"1","volume":"17","author":"S Diamond","year":"2016","unstructured":"Diamond S, Boyd S (2016) CVXPY: a python-embedded modeling language for convex optimization. J Mach Learn Res 17(83):1\u20135","journal-title":"J Mach Learn Res"},{"key":"2845_CR8","doi-asserted-by":"crossref","unstructured":"Ferdinand N, Gharachorloo B, Draper SC (2017) Anytime exploitation of stragglers in synchronous stochastic gradient descent. In: 2017 16th IEEE International Conference on Machine Learning and Applications (ICMLA), pp 141\u2013146","DOI":"10.1109\/ICMLA.2017.0-166"},{"key":"2845_CR9","unstructured":"Google. grpc"},{"key":"2845_CR10","unstructured":"Goyal P, Doll\u00e1r P, Girshick R, Noordhuis P, Wesolowski L, Kyrola A, Tulloch A, Jia Y, He K (2017) Accurate, large minibatch SGD: training imagenet in 1 hour. arXiv preprint \narXiv:1706.02677"},{"key":"2845_CR11","doi-asserted-by":"crossref","unstructured":"Harlap A, Cui H, Dai W, Wei J, Ganger GR, Gibbons PB, Gibson GA, Xing EP (2016) Addressing the straggler problem for iterative convergent parallel ML. In: Proceedings of the Seventh ACM Symposium on Cloud Computing, SoCC\u201916, New York, NY, USA. ACM, pp 98\u2013111","DOI":"10.1145\/2987550.2987554"},{"key":"2845_CR12","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"2845_CR13","doi-asserted-by":"crossref","unstructured":"Iandola FN, Moskewicz MW, Ashraf K, Keutzer K (2016) Firecaffe: near-linear acceleration of deep neural network training on compute clusters. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2592\u20132600","DOI":"10.1109\/CVPR.2016.284"},{"key":"2845_CR14","unstructured":"Jeon M, Venkataraman S, Qian J, Phanishayee A, Xiao W, Yang F (2018) Multi-tenant GPU clusters for deep learning workloads: analysis and implications. Technical report"},{"key":"2845_CR15","doi-asserted-by":"crossref","unstructured":"Jiang J, Cui B, Zhang C, Yu L (2017) Heterogeneity-aware distributed parameter servers. In: Proceedings of the 2017 ACM International Conference on Management of Data, SIGMOD\u201917, New York, NY, USA. ACM, pp 463\u2013478","DOI":"10.1145\/3035918.3035933"},{"key":"2845_CR16","unstructured":"Krizhevsky A, Hinton G (2009) Learning multiple layers of features from tiny images. Technical report, Citeseer"},{"key":"2845_CR17","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. In: Advances in neural information processing systems, pp 1097\u20131105"},{"issue":"7553","key":"2845_CR18","doi-asserted-by":"publisher","first-page":"436","DOI":"10.1038\/nature14539","volume":"521","author":"Y LeCun","year":"2015","unstructured":"LeCun Y, Bengio Y, Hinton G (2015) Deep learning. Nature 521(7553):436","journal-title":"Nature"},{"issue":"11","key":"2845_CR19","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun Y, Bottou L, Bengio Y, Haffner P (1998) Gradient-based learning applied to document recognition. Proc IEEE 86(11):2278\u20132324","journal-title":"Proc IEEE"},{"issue":"239","key":"2845_CR20","first-page":"2","volume":"2014","author":"D Merkel","year":"2014","unstructured":"Merkel D (2014) Docker: lightweight linux containers for consistent development and deployment. Linux J 2014(239):2","journal-title":"Linux J"},{"key":"2845_CR21","doi-asserted-by":"crossref","unstructured":"Robbins H, Monro S (1951) A stochastic approximation method. The annals of mathematical statistics, pp 400\u2013407","DOI":"10.1214\/aoms\/1177729586"},{"key":"2845_CR22","unstructured":"Tandon R, Lei Q, Dimakis AG, Karampatziakis N (2017) Gradient coding: avoiding stragglers in distributed learning. In: Precup D, Teh YW (eds) Proceedings of the 34th International Conference on Machine Learning, volume\u00a070 of Proceedings of Machine Learning Research, PMLR. International Convention Centre, Sydney, Australia, 06\u201311 Aug 2017, pp 3368\u20133376"},{"key":"2845_CR23","doi-asserted-by":"crossref","unstructured":"Yan F, Ruwase O, He Y, Chilimbi T (2015) Performance modeling and scalability optimization of distributed deep learning systems. In: Proceedings of the 21th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining. ACM, pp 1355\u20131364","DOI":"10.1145\/2783258.2783270"},{"key":"2845_CR24","doi-asserted-by":"crossref","unstructured":"Yang E, Kim S, Kim T, Jeon M, Park S, Youn C (2018) An adaptive batch-orchestration algorithm for the heterogeneous GPU cluster environment in distributed deep learning system. In: 2018 IEEE International Conference on Big Data and Smart Computing (BigComp), pp 725\u2013728","DOI":"10.1109\/BigComp.2018.00136"},{"issue":"12","key":"2845_CR25","doi-asserted-by":"publisher","first-page":"1283","DOI":"10.14778\/2732977.2733001","volume":"7","author":"C Zhang","year":"2014","unstructured":"Zhang C, R\u00e9 C (2014) Dimmwitted: a study of main-memory statistical analytics. Proc VLDB Endow 7(12):1283\u20131294","journal-title":"Proc VLDB Endow"},{"key":"2845_CR26","unstructured":"Zinkevich M, Weimer M, Li L, Smola AJ (2010) Parallelized stochastic gradient descent. In: Advances in neural information processing systems, pp 2595\u20132603"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-019-02845-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11227-019-02845-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-019-02845-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,5,19]],"date-time":"2020-05-19T09:34:47Z","timestamp":1589880887000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11227-019-02845-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,4,16]]},"references-count":26,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2020,1]]}},"alternative-id":["2845"],"URL":"https:\/\/doi.org\/10.1007\/s11227-019-02845-2","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,4,16]]},"assertion":[{"value":"16 April 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}