{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T20:38:48Z","timestamp":1771706328089,"version":"3.50.1"},"reference-count":75,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,9,1]],"date-time":"2019-09-01T00:00:00Z","timestamp":1567296000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,9,1]],"date-time":"2019-09-01T00:00:00Z","timestamp":1567296000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,9,1]],"date-time":"2019-09-01T00:00:00Z","timestamp":1567296000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,9]]},"DOI":"10.1109\/cluster.2019.8891023","type":"proceedings-article","created":{"date-parts":[[2019,11,13]],"date-time":"2019-11-13T13:00:53Z","timestamp":1573650053000},"page":"1-12","source":"Crossref","is-referenced-by-count":33,"title":["Efficient User-Level Storage Disaggregation for Deep Learning"],"prefix":"10.1109","author":[{"given":"Yue","family":"Zhu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weikuan","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bing","family":"Jiao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kathryn","family":"Mohror","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Adam","family":"Moody","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fahim","family":"Chowdhury","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1145\/2834976.2834984"},{"key":"ref72","article-title":"FanStore: Enabling Efficient and Scalable I\/O for Distributed Deep Learning","author":"zhang","year":"2018"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/CloudCom.2017.14"},{"key":"ref70","article-title":"Yet Another Accelerated SGD: ResNet-50 Training on ImageNet in 74.7 seconds","author":"yamazaki","year":"2019"},{"key":"ref74","author":"zhu","year":"0","journal-title":"Multi-Client DeepIO for Large-Scale Deep Learning on HPC Systems"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750392"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/MASCOTS.2018.00023"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1145\/2647868.2654889"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/3239563"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/3078468.3078483"},{"key":"ref30","article-title":"Accurate, Large Minibatch SGD: Training Imagenet in 1 Hour","author":"goyal","year":"2017"},{"key":"ref37","article-title":"Highly Scalable Deep Learning Training System with Mixed-Precision: Training Imagenet in Four Minutes","author":"jia","year":"2018"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/2925426.2926290"},{"key":"ref35","year":"0","journal-title":"Intel RSA"},{"key":"ref34","year":"0","journal-title":"Moonshot System The Worlds First Software-Defined Servers"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/MC.2013.222"},{"key":"ref62","first-page":"1","article-title":"Facebook&#x2019;s data center infrastructure: Open compute, disaggregated rack, and beyond","author":"taylor","year":"2015","journal-title":"Optical Fiber Communication Conference and Exhibition (OFC) 2015"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/MASCOTS.2017.35"},{"key":"ref63","first-page":"5","article-title":"LBANN: Livermore Big Artificial Neural Network HPC Toolkit","author":"van essen","year":"2015","journal-title":"Proceedings of the Workshop on Machine Learning in High-Performance Computing Environments"},{"key":"ref28","first-page":"249","article-title":"Network Requirements for Resource Disaggregation","volume":"16","author":"gao","year":"2016","journal-title":"OSDI"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER.2018.00049"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CCGRID.2017.49"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2017.39"},{"key":"ref66","first-page":"69","article-title":"An Ephemeral Burst-Buffer File System for Scientific Applications","author":"wang","year":"2016","journal-title":"Proceedings of the International Conference for High Performance Computing Networking Storage and Analysis"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-15277-1_37"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/BigData.2014.7004215"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126923"},{"key":"ref69","article-title":"Performance Analysis of Containerized Applications on Local and Remote Storage","author":"xu","year":"2017","journal-title":"Proc of MSST"},{"key":"ref2","year":"0","journal-title":"Caffe Database Layer"},{"key":"ref1","year":"0","journal-title":"An overview of gradient descent optimization algorithms"},{"key":"ref20","article-title":"Firebox: A Hardware Building Block for 2020 Warehouse-Scale Computers","volume":"13","author":"asanovic","year":"2014","journal-title":"USENIX FAST"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.14778\/3229863.3229872"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.2200\/S00516ED2V01Y201306CAC024"},{"key":"ref24","author":"culler","year":"1998","journal-title":"Parallel Computer Architecture A Hardware\/Software Approach"},{"key":"ref23","first-page":"80","article-title":"I\/O Characterization and Performance Evaluation of BeeGFS for Deep Learning","author":"chowdhury","year":"2019","journal-title":"Proceedings of the 48th International Conference on Parallel Processing"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/HiPC.2016.026"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/SC.Companion.2012.146"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1145\/2654822.2541959"},{"key":"ref59","first-page":"38","article-title":"Crail: A High-Performance I\/O Architecture for Distributed Data Processing","volume":"40","author":"stuedi","year":"2017","journal-title":"IEEE Data Eng Bull"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1145\/2287036.2287050"},{"key":"ref57","article-title":"Don&#x2019;t Decay the Learning Rate, Increase the Batch Size","author":"smith","year":"2017"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1145\/3079079.3079087"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid.2014.24"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/HPCC-SmartCity-DSS.2017.29"},{"key":"ref53","first-page":"720","article-title":"Parallel I\/O Optimizations for Scalable Deep Learning","author":"pumma","year":"2017","journal-title":"IEEE International Conference on Parallel and Distributed Systems (ICPADS)"},{"key":"ref52","doi-asserted-by":"crossref","first-page":"574","DOI":"10.1016\/j.parco.2014.09.011","article-title":"A Complete and Efficient CUDA-Sharing Solution for HPC Clusters","volume":"40","author":"pe\u00f1a","year":"2014","journal-title":"Parallel Computing"},{"key":"ref10","year":"0","journal-title":"Removing The Storage Bottleneck for AI"},{"key":"ref11","year":"0","journal-title":"SeaMicro SM10000 System Overview"},{"key":"ref40","first-page":"690","article-title":"Rack-Scale Disaggregated Cloud Data Centers: The dReDBox Project Vision","author":"k katrinis","year":"2016","journal-title":"Design Automation Test in Europe Conference Exhibition (DATE)"},{"key":"ref12","year":"0","journal-title":"Sierra"},{"key":"ref13","year":"0","journal-title":"SPDK NVMe over Fabrics Target"},{"key":"ref14","year":"0","journal-title":"SPDK User Space Drivers"},{"key":"ref15","year":"0","journal-title":"SUMMIT"},{"key":"ref16","year":"0","journal-title":"TensorFlow Adding a New Op"},{"key":"ref17","year":"0","journal-title":"TensorFlow Dataset API tf data FixedLengthRecordDataset shuffle()"},{"key":"ref18","article-title":"Tensorflow: Large-Scale Machine Learning on Heterogeneous Distributed Systems","author":"abadi","year":"2016"},{"key":"ref19","article-title":"Extremely Large Minibatch SGD: Training Resnet-50 on Imagenet in 15 Minutes","author":"akiba","year":"2017"},{"key":"ref4","year":"0","journal-title":"DataWarp"},{"key":"ref3","year":"0","journal-title":"Data Storage Keeping Pace for AI and Deep Learning"},{"key":"ref6","year":"0","journal-title":"Infinite Memory Engine (IME)"},{"key":"ref5","year":"0","journal-title":"EMC ABBA"},{"key":"ref8","year":"0","journal-title":"Introducing data center fabric the next-generation facebook data center network"},{"key":"ref7","year":"0","journal-title":"Intel rack scale design"},{"key":"ref49","first-page":"17","article-title":"Decibel: Isolation and Sharing in Disaggregated Rack-Scale Storage","author":"nanavati","year":"2017","journal-title":"NSDI"},{"key":"ref9","year":"0","journal-title":"NVMe over fabric"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/HPCC-SmartCity-DSS.2016.0157"},{"key":"ref45","first-page":"773","article-title":"Octopus: an RDMA-enabled Distributed Persistent Memory File System","author":"lu","year":"2017","journal-title":"2017 USENIX Annual Technical Conference (USENIX ATC 17)"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1145\/3230543.3230557"},{"key":"ref47","article-title":"ImageNet\/ResNet-50 Training in 224 Seconds","author":"mikami","year":"2018"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1145\/3093315.3037732"},{"key":"ref41","first-page":"29","article-title":"Flash Storage Disaggregation","author":"klimovic","year":"2016","journal-title":"Proceedings of the Eleventh European Conference on Computer Systems"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/MSST.2012.6232369"},{"key":"ref43","article-title":"Understanding Rack-Scale Disaggregated Storage","author":"legtchenko","year":"2017","journal-title":"USENIX Workshop on Hot Topics in Storage and File Systems (HotStorage)"}],"event":{"name":"2019 IEEE International Conference on Cluster Computing (CLUSTER)","location":"Albuquerque, NM, USA","start":{"date-parts":[[2019,9,23]]},"end":{"date-parts":[[2019,9,26]]}},"container-title":["2019 IEEE International Conference on Cluster Computing (CLUSTER)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8884608\/8890988\/08891023.pdf?arnumber=8891023","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,17]],"date-time":"2022-07-17T17:49:07Z","timestamp":1658080147000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8891023\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,9]]},"references-count":75,"URL":"https:\/\/doi.org\/10.1109\/cluster.2019.8891023","relation":{},"subject":[],"published":{"date-parts":[[2019,9]]}}}