{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T00:12:40Z","timestamp":1787011960730,"version":"build-2736575974"},"reference-count":79,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,9,1]],"date-time":"2019-09-01T00:00:00Z","timestamp":1567296000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,9,1]],"date-time":"2019-09-01T00:00:00Z","timestamp":1567296000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,9,1]],"date-time":"2019-09-01T00:00:00Z","timestamp":1567296000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,9]]},"DOI":"10.1109\/cluster.2019.8890993","type":"proceedings-article","created":{"date-parts":[[2019,11,13]],"date-time":"2019-11-13T13:00:53Z","timestamp":1573650053000},"page":"1-12","source":"Crossref","is-referenced-by-count":14,"title":["A Quantitative Study of Deep Learning Training on Heterogeneous Supercomputers"],"prefix":"10.1109","author":[{"given":"Jingoo","family":"Han","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Luna","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"M. Mustafa","family":"Rafique","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ali R.","family":"Butt","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Seung-Hwan","family":"Lim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref73","year":"2018","journal-title":"Lightning memory-mapped database manager (LMDB)"},{"key":"ref72","first-page":"11","article-title":"An analysis of image storage systems for scalable training of deep neural networks","volume":"5","author":"lim","year":"2016","journal-title":"System"},{"key":"ref71","first-page":"583","article-title":"Scaling distributed machine learning with the parameter server","author":"li","year":"2014","journal-title":"11th USENIX Symposium on Operating Systems Design and Implementation (OSDI 14)"},{"key":"ref70","first-page":"2","article-title":"Parameter server for distributed machine learning","volume":"6","author":"li","year":"2013","journal-title":"NIPS Big Learning Workshop"},{"key":"ref76","author":"xu","year":"2019","journal-title":"A Workload-aware Resource Management and Scheduling System for Big Data Analysis"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1109\/PDSW-DISCS.2018.00011"},{"key":"ref74","year":"2018","journal-title":"TFRecords"},{"key":"ref39","article-title":"Gpfs: A shared-disk file system for large computing clusters","volume":"2","author":"schmuck","year":"2002","journal-title":"FAST"},{"key":"ref75","article-title":"Extremely large minibatch sgd: training resnet-50 on imagenet in 15 minutes","author":"akiba","year":"2017"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00055"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1109\/ICDM.2016.0028"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1145\/3267809.3267840"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref32","year":"2019","journal-title":"SUMMIT"},{"key":"ref31","author":"xu","year":"2018","journal-title":"Deep learning at scale on nvidia v100 accelerators"},{"key":"ref30","article-title":"Scaling grpc tensorflow on 512 nodes of cori supercomputer","author":"mathuriya","year":"2017"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2015.2442980"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/1296907.1296909"},{"key":"ref35","article-title":"Scalability in the xfs file system","volume":"15","author":"sweeney","year":"1996","journal-title":"USENIX Annual Technical Conference"},{"key":"ref34","year":"2018","journal-title":"Lustre"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126933"},{"key":"ref62","year":"2018","journal-title":"gRPC"},{"key":"ref61","year":"2018","journal-title":"Google Protocol Buffers"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1145\/103162.103163"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CCGRID.2017.110"},{"key":"ref64","article-title":"Highly scalable deep learning training system with mixed-precision: Training imagenet in four minutes","author":"jia","year":"2018"},{"key":"ref27","first-page":"265","article-title":"Tensorflow: a system for large-scale machine learning","volume":"16","author":"abadi","year":"2016","journal-title":"OSDI"},{"key":"ref65","article-title":"Accurate, large minibatch sgd: training imagenet in 1 hour","author":"goyal","year":"2017"},{"key":"ref66","first-page":"1:1","article-title":"Imagenet training in minutes","author":"you","year":"2018","journal-title":"Proceedings of the 47th International Conference on Parallel Processing ICPP 2018"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS.2017.259"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00068"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126912"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1145\/3035918.3035933"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/BigData.2017.8257935"},{"key":"ref1","doi-asserted-by":"crossref","first-page":"436","DOI":"10.1038\/nature14539","article-title":"Deep learning","volume":"521","author":"lecun","year":"2015","journal-title":"Nature"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126916"},{"key":"ref22","article-title":"Hoard: A distributed data caching system to accelerate deep learning training on the cloud","author":"pinto","year":"2018"},{"key":"ref21","doi-asserted-by":"crossref","first-page":"18","DOI":"10.1007\/978-3-319-69179-4_2","article-title":"Distributed training large-scale deep architectures","author":"zou","year":"2017","journal-title":"International Conference on Advanced Data Mining and Applications"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2017.40"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/2987550.2987586"},{"key":"ref26","article-title":"One weird trick for parallelizing convolutional neural networks","author":"krizhevsky","year":"2014"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2017.26"},{"key":"ref50","article-title":"Scaling sgd batch size to 32k for imagenet training","author":"you","year":"2017"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICOIN.2018.8343173"},{"key":"ref59","year":"2019","journal-title":"Unix dd"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2018.8573521"},{"key":"ref57","year":"2018","journal-title":"Tensorflow-slim"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-018-6042-1"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-75025-5_10"},{"key":"ref54","article-title":"cudnn: Efficient primitives for deep learning","author":"chetlur","year":"2014"},{"key":"ref53","year":"2019","journal-title":"Compute Unified Device Architecture (CUDA)"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/ISSREW.2018.00024"},{"key":"ref10","year":"2018","journal-title":"Top500"},{"key":"ref11","year":"2018","journal-title":"NVIDIA Tesla V100"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126932"},{"key":"ref12","year":"2018","journal-title":"NVlink"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/MCSE.2018.021651341"},{"key":"ref14","year":"2018","journal-title":"Sierra"},{"key":"ref15","year":"2018","journal-title":"AI Bridging Cloud Infrastructure (ABCI)"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00054"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/MSST.2012.6232369"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/MCSE.2015.4"},{"key":"ref19","year":"2018","journal-title":"InfiniBand Trade Association"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1093\/mnras\/stv632"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/PDSW-DISCS.2016.009"},{"key":"ref6","article-title":"Deep supervised and convolutional generative stochastic network for protein secondary structure prediction","author":"zhou","year":"2014"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2017.07.005"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.2172\/1473756"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1177\/1094342017690910"},{"key":"ref49","article-title":"Modeling and evaluation of synchronous stochastic gradient descent in distributed deep learning on multiple gpus","author":"shi","year":"2018"},{"key":"ref9","year":"2018","journal-title":"ORNL Launches Summit Supercomputer"},{"key":"ref46","first-page":"571","article-title":"Project adam: Building an efficient and scalable deep learning training system","volume":"14","author":"chilimbi","year":"2014","journal-title":"OSDI"},{"key":"ref45","first-page":"1337","article-title":"Deep learning with cots hpc systems","author":"coates","year":"2013","journal-title":"International Conference on Machine Learning"},{"key":"ref48","first-page":"1223","article-title":"Large scale distributed deep networks","author":"dean","year":"2012","journal-title":"Advances in neural information processing systems"},{"key":"ref47","article-title":"Multi-gpu training of convnets","author":"yadan","year":"2013"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/5.726791"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CCGRID.2007.51"},{"key":"ref44","article-title":"Towards evolutional compression","author":"wang","year":"2017"},{"key":"ref43","first-page":"1135","article-title":"Learning both weights and connections for efficient neural network","author":"han","year":"2015","journal-title":"Advances in neural information processing systems"}],"event":{"name":"2019 IEEE International Conference on Cluster Computing (CLUSTER)","location":"Albuquerque, NM, USA","start":{"date-parts":[[2019,9,23]]},"end":{"date-parts":[[2019,9,26]]}},"container-title":["2019 IEEE International Conference on Cluster Computing (CLUSTER)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8884608\/8890988\/08890993.pdf?arnumber=8890993","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,17]],"date-time":"2022-07-17T17:55:12Z","timestamp":1658080512000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8890993\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,9]]},"references-count":79,"URL":"https:\/\/doi.org\/10.1109\/cluster.2019.8890993","relation":{},"subject":[],"published":{"date-parts":[[2019,9]]}}}