{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T01:08:51Z","timestamp":1780708131536,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,8,5]],"date-time":"2019-08-05T00:00:00Z","timestamp":1564963200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,8,5]]},"DOI":"10.1145\/3337821.3337892","type":"proceedings-article","created":{"date-parts":[[2019,7,25]],"date-time":"2019-07-25T12:34:36Z","timestamp":1564058076000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":17,"title":["DLBooster"],"prefix":"10.1145","author":[{"given":"Yang","family":"Cheng","sequence":"first","affiliation":[{"name":"Tsinghua University Microsoft Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dan","family":"Li","sequence":"additional","affiliation":[{"name":"Tsinghua University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhiyuan","family":"Guo","sequence":"additional","affiliation":[{"name":"Microsoft Research Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Binyao","family":"Jiang","sequence":"additional","affiliation":[{"name":"Microsoft Research Shanghai Jiao Tong University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiaxin","family":"Lin","sequence":"additional","affiliation":[{"name":"Microsoft Research Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xi","family":"Fan","sequence":"additional","affiliation":[{"name":"Microsoft Research Shanghai Jiao Tong University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinkun","family":"Geng","sequence":"additional","affiliation":[{"name":"Tsinghua University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinyi","family":"Yu","sequence":"additional","affiliation":[{"name":"Microsoft Research Shanghai Jiao Tong University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Bai","sequence":"additional","affiliation":[{"name":"Microsoft Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lei","family":"Qu","sequence":"additional","affiliation":[{"name":"Microsoft Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ran","family":"Shu","sequence":"additional","affiliation":[{"name":"Microsoft Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Cheng","sequence":"additional","affiliation":[{"name":"Microsoft Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongqiang","family":"Xiong","sequence":"additional","affiliation":[{"name":"Microsoft Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianping","family":"Wu","sequence":"additional","affiliation":[{"name":"Tsinghua University"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2019,8,5]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Tensorflow: A system for large-scale machine learning. In 12th {USENIX} Symposium on Operating Systems Design and Implementation ({OSDI} 16). 265--283.","author":"Mart\u00edn Abadi","year":"2016","unstructured":"Mart\u00edn Abadi et al. 2016 . Tensorflow: A system for large-scale machine learning. In 12th {USENIX} Symposium on Operating Systems Design and Implementation ({OSDI} 16). 265--283. Mart\u00edn Abadi et al. 2016. Tensorflow: A system for large-scale machine learning. In 12th {USENIX} Symposium on Operating Systems Design and Implementation ({OSDI} 16). 265--283."},{"key":"e_1_3_2_1_2_1","unstructured":"Anirudh Acharya et al. 2018. Image Transforms and RecordIO file Creation of MXNet. https:\/\/cwiki.apache.org\/confluence\/display\/MXNET\/Image+Transforms+and+RecordIO+file+Creation.  Anirudh Acharya et al. 2018. Image Transforms and RecordIO file Creation of MXNet. https:\/\/cwiki.apache.org\/confluence\/display\/MXNET\/Image+Transforms+and+RecordIO+file+Creation."},{"key":"e_1_3_2_1_3_1","unstructured":"Takuya Akiba et al. 2017. Extremely large minibatch sgd: Training resnet-50 on imagenet in 15 minutes. arXiv preprint arXiv:1711.04325 (2017).  Takuya Akiba et al. 2017. Extremely large minibatch sgd: Training resnet-50 on imagenet in 15 minutes. arXiv preprint arXiv:1711.04325 (2017)."},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of COMPSTAT'2010","author":"L\u00e9on","unstructured":"L\u00e9on Bottou et al. 2010. Large-scale machine learning with stochastic gradient descent . In Proceedings of COMPSTAT'2010 . Springer, 177--186. L\u00e9on Bottou et al. 2010. Large-scale machine learning with stochastic gradient descent. In Proceedings of COMPSTAT'2010. Springer, 177--186."},{"key":"e_1_3_2_1_5_1","unstructured":"Jianmin Chen et al. 2016. Revisiting distributed synchronous SGD. arXiv preprint arXiv:1604.00981 (2016).  Jianmin Chen et al. 2016. Revisiting distributed synchronous SGD. arXiv preprint arXiv:1604.00981 (2016)."},{"key":"e_1_3_2_1_6_1","volume-title":"Mxnet: A flexible and efficient machine learning library for heterogeneous distributed systems. arXiv:1512.01274","author":"Tianqi Chen","year":"2015","unstructured":"Tianqi Chen et al. 2015 . Mxnet: A flexible and efficient machine learning library for heterogeneous distributed systems. arXiv:1512.01274 (2015). Tianqi Chen et al. 2015. Mxnet: A flexible and efficient machine learning library for heterogeneous distributed systems. arXiv:1512.01274 (2015)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/s42045-018-0009-7"},{"key":"e_1_3_2_1_8_1","unstructured":"Minsik Cho et al. 2017. PowerAI DDL. arXiv preprint arXiv:1708.02188 (2017).  Minsik Cho et al. 2017. PowerAI DDL. arXiv preprint arXiv:1708.02188 (2017)."},{"key":"e_1_3_2_1_9_1","unstructured":"Intel Corporation. 2018. A brief JPEG decoder example design on Intel Stritax V A7 FPGA. https:\/\/www.intel.com\/content\/dam\/www\/programmable\/us\/en\/pdfs\/support\/examples\/download\/exm_opencl_jpegdecoder.pdf.  Intel Corporation. 2018. A brief JPEG decoder example design on Intel Stritax V A7 FPGA. https:\/\/www.intel.com\/content\/dam\/www\/programmable\/us\/en\/pdfs\/support\/examples\/download\/exm_opencl_jpegdecoder.pdf."},{"key":"e_1_3_2_1_10_1","unstructured":"Intel Corporation. 2018. Intel Arria-10 FPGAs. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/programmable\/fpga\/arria-10.html.  Intel Corporation. 2018. Intel Arria-10 FPGAs. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/programmable\/fpga\/arria-10.html."},{"key":"e_1_3_2_1_11_1","unstructured":"Intel Corporation. 2019. A brief Introduction to the Intel Optane SSD 900p Series. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/memory-storage\/solid-state-drives\/gaming-enthusiast-ssds\/optane-900p-series.html.  Intel Corporation. 2019. A brief Introduction to the Intel Optane SSD 900p Series. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/memory-storage\/solid-state-drives\/gaming-enthusiast-ssds\/optane-900p-series.html."},{"key":"e_1_3_2_1_12_1","unstructured":"NVIDIA Corporation. 2018. Announcing NVIDIA DALI and NVIDIA nvJPEG. https:\/\/news.developer.nvidia.com\/announcing-nvidia-dali-and-nvidia-nvjpeg\/.  NVIDIA Corporation. 2018. Announcing NVIDIA DALI and NVIDIA nvJPEG. https:\/\/news.developer.nvidia.com\/announcing-nvidia-dali-and-nvidia-nvjpeg\/."},{"key":"e_1_3_2_1_13_1","unstructured":"NVIDIA Corporation. 2018. NVCaffe an NVIDIA-maintained fork of Berkeley Vision and Learning Center (BVLC) Caffe tuned for NVIDIA GPUs. https:\/\/www.nvidia.com\/en-us\/data-center\/gpu-accelerated-applications\/caffe\/.  NVIDIA Corporation. 2018. NVCaffe an NVIDIA-maintained fork of Berkeley Vision and Learning Center (BVLC) Caffe tuned for NVIDIA GPUs. https:\/\/www.nvidia.com\/en-us\/data-center\/gpu-accelerated-applications\/caffe\/."},{"key":"e_1_3_2_1_14_1","unstructured":"NVIDIA Corporation. 2018. NVIDIA DGX-2: the world's most powerful AI system for the most complex AI challenges. https:\/\/www.nvidia.com\/en-us\/data-center\/dgx-2\/.  NVIDIA Corporation. 2018. NVIDIA DGX-2: the world's most powerful AI system for the most complex AI challenges. https:\/\/www.nvidia.com\/en-us\/data-center\/dgx-2\/."},{"key":"e_1_3_2_1_15_1","unstructured":"NVIDIA Corporation. 2018. NVIDIA TensorRT Programmable Inference Accelerator. https:\/\/developer.nvidia.com\/tensorrt.  NVIDIA Corporation. 2018. NVIDIA TensorRT Programmable Inference Accelerator. https:\/\/developer.nvidia.com\/tensorrt."},{"key":"e_1_3_2_1_16_1","unstructured":"NVIDIA Corporation. 2018. nvJPEG GPU-accelerated JPEG decoder. https:\/\/developer.nvidia.com\/nvjpeg.  NVIDIA Corporation. 2018. nvJPEG GPU-accelerated JPEG decoder. https:\/\/developer.nvidia.com\/nvjpeg."},{"key":"e_1_3_2_1_17_1","unstructured":"Mark Daoust et al. 2018. Using TFRecords and tf.Example in TensorFlow. https:\/\/www.tensorflow.org\/tutorials\/load_data\/tf-records.  Mark Daoust et al. 2018. Using TFRecords and tf.Example in TensorFlow. https:\/\/www.tensorflow.org\/tutorials\/load_data\/tf-records."},{"key":"e_1_3_2_1_18_1","first-page":"64","article-title":"Recent advances in deep learning for speech research at Microsoft","volume":"26","author":"Ltsc Deng","year":"2013","unstructured":"Ltsc Deng et al. 2013 . Recent advances in deep learning for speech research at Microsoft .. In ICASSP , Vol. 26. 64 . Ltsc Deng et al. 2013. Recent advances in deep learning for speech research at Microsoft.. In ICASSP, Vol. 26. 64.","journal-title":"ICASSP"},{"key":"e_1_3_2_1_19_1","volume-title":"Proceedings of NSDI'18","author":"Daniel","unstructured":"Daniel Firestone et al. 2018. Azure Accelerated Networking: SmartNICs in the Public Cloud . In Proceedings of NSDI'18 , Renton, WA. Daniel Firestone et al. 2018. Azure Accelerated Networking: SmartNICs in the Public Cloud. In Proceedings of NSDI'18, Renton, WA."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"Robin Flowerdew et al. 1991. Using areal interpolation methods in geographic information systems. Papers in regional science 70 3 (1991) 303--315.  Robin Flowerdew et al. 1991. Using areal interpolation methods in geographic information systems. Papers in regional science 70 3 (1991) 303--315.","DOI":"10.1007\/BF01434424"},{"key":"e_1_3_2_1_21_1","volume-title":"Proc. IEEE ICASSP","author":"Jort","year":"2017","unstructured":"Jort F. Gemmeke et al. 2017. Audio Set: An ontology and human-labeled dataset for audio events . In Proc. IEEE ICASSP 2017 . New Orleans, LA. Jort F. Gemmeke et al. 2017. Audio Set: An ontology and human-labeled dataset for audio events. In Proc. IEEE ICASSP 2017. New Orleans, LA."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3229543.3229544"},{"key":"e_1_3_2_1_23_1","volume-title":"ICDE'19","author":"Jinkun","unstructured":"Jinkun Geng et al. 2019. Rima:An RDMA-Accelerated Model-Parallelized Solution to Large-Scale Matrix Factorization . In ICDE'19 . Jinkun Geng et al. 2019. Rima:An RDMA-Accelerated Model-Parallelized Solution to Large-Scale Matrix Factorization. In ICDE'19."},{"key":"e_1_3_2_1_24_1","unstructured":"Priya Goyal et al. 2017. Accurate large minibatch SGD: training imagenet in 1 hour. arXiv preprint arXiv:1706.02677 (2017).  Priya Goyal et al. 2017. Accurate large minibatch SGD: training imagenet in 1 hour. arXiv preprint arXiv:1706.02677 (2017)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/2934872.2934908"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"e_1_3_2_1_27_1","volume-title":"IEEE CVPR'18","author":"Kaiming","unstructured":"Kaiming He et al. 2016. Deep residual learning for image recognition . In IEEE CVPR'18 . 770--778. Kaiming He et al. 2016. Deep residual learning for image recognition. In IEEE CVPR'18. 770--778."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3106989.3106997"},{"key":"e_1_3_2_1_29_1","unstructured":"M. Jaderberg et al. 2014. Reading Text in the Wild with Convolutional Neural Networks. arXiv preprint arXiv:1412.1842 (2014).  M. Jaderberg et al. 2014. Reading Text in the Wild with Convolutional Neural Networks. arXiv preprint arXiv:1412.1842 (2014)."},{"key":"e_1_3_2_1_30_1","unstructured":"Xianyan Jia et al. 2018. Highly Scalable Deep Learning Training System with Mixed-Precision: Training ImageNet in Four Minutes. arXiv preprint arXiv:1807.11205 (2018).  Xianyan Jia et al. 2018. Highly Scalable Deep Learning Training System with Mixed-Precision: Training ImageNet in Four Minutes. arXiv preprint arXiv:1807.11205 (2018)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/2647868.2654889"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3322795.3331463"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3322795.3331461"},{"key":"e_1_3_2_1_34_1","volume-title":"Deep Learning with Python","author":"Ketkar Nikhil","unstructured":"Nikhil Ketkar . 2017. Introduction to pytorch . In Deep Learning with Python . Springer , 195--208. Nikhil Ketkar. 2017. Introduction to pytorch. In Deep Learning with Python. Springer, 195--208."},{"key":"e_1_3_2_1_35_1","volume-title":"NIPS","author":"Alex","year":"2012","unstructured":"Alex Krizhevsky et al. 2012. Imagenet classification with deep convolutional neural networks . In NIPS 2012 . 1097--1105. Alex Krizhevsky et al. 2012. Imagenet classification with deep convolutional neural networks. In NIPS 2012. 1097--1105."},{"key":"e_1_3_2_1_36_1","volume-title":"Speech and Signal Processing (ICASSP), 2013 IEEE International Conference on. IEEE, 8595--8598","author":"Le Quoc V","year":"2013","unstructured":"Quoc V Le . 2013 . Building high-level features using large scale unsupervised learning. In Acoustics , Speech and Signal Processing (ICASSP), 2013 IEEE International Conference on. IEEE, 8595--8598 . Quoc V Le. 2013. Building high-level features using large scale unsupervised learning. In Acoustics, Speech and Signal Processing (ICASSP), 2013 IEEE International Conference on. IEEE, 8595--8598."},{"key":"e_1_3_2_1_37_1","volume-title":"International conference on artificial neural networks","volume":"60","author":"Yann","unstructured":"Yann LeCun et al. 1995. Comparison of learning algorithms for handwritten digit recognition . In International conference on artificial neural networks , Vol. 60 . Perth, Australia, 53--60. Yann LeCun et al. 1995. Comparison of learning algorithms for handwritten digit recognition. In International conference on artificial neural networks, Vol. 60. Perth, Australia, 53--60."},{"key":"e_1_3_2_1_38_1","unstructured":"Yann LeCun etal 1998. THE MNIST DATABASE of handwritten digits. Retrieved 2012 from http:\/\/yann.lecun.com\/exdb\/mnist\/  Yann LeCun et al. 1998. THE MNIST DATABASE of handwritten digits. Retrieved 2012 from http:\/\/yann.lecun.com\/exdb\/mnist\/"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"crossref","unstructured":"Yann LeCun etal 2015. Deep learning. nature 521 7553 (2015) 436.  Yann LeCun et al. 2015. Deep learning. nature 521 7553 (2015) 436.","DOI":"10.1038\/nature14539"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2018.00023"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_42_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman . 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 ( 2014 ). Karen Simonyan and Andrew Zisserman. 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","unstructured":"John E Stone etal 2010. OpenCL: A parallel programming standard for heterogeneous computing systems. Computing in science & engineering 12 3 (2010) 66--73.   John E Stone et al. 2010. OpenCL: A parallel programming standard for heterogeneous computing systems. Computing in science & engineering 12 3 (2010) 66--73.","DOI":"10.1109\/MCSE.2010.69"},{"key":"e_1_3_2_1_44_1","volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition. 1--9.","author":"Christian","unstructured":"Christian Szegedy et al. 2015. Going deeper with convolutions . In Proceedings of the IEEE conference on computer vision and pattern recognition. 1--9. Christian Szegedy et al. 2015. Going deeper with convolutions. In Proceedings of the IEEE conference on computer vision and pattern recognition. 1--9."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3109859.3109900"},{"key":"e_1_3_2_1_46_1","volume-title":"Impact of Network Topology on the Performance of DML: Theoretical Analysis and Practical Factors. In IEEE INFOCOM","author":"Shuai","year":"2019","unstructured":"Shuai Wang et al. 2019 . Impact of Network Topology on the Performance of DML: Theoretical Analysis and Practical Factors. In IEEE INFOCOM 2019 . Shuai Wang et al. 2019. Impact of Network Topology on the Performance of DML: Theoretical Analysis and Practical Factors. In IEEE INFOCOM 2019."},{"key":"e_1_3_2_1_47_1","unstructured":"Yang You et al. 2017. ImageNet training in minutes. CoRR abs\/1709.05011 (2017).  Yang You et al. 2017. ImageNet training in minutes. CoRR abs\/1709.05011 (2017)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/2785956.2787484"},{"key":"e_1_3_2_1_49_1","unstructured":"Martin Zinkevich et al. 2010. Parallelized stochastic gradient descent. In Advances in neural information processing systems. 2595--2603.   Martin Zinkevich et al. 2010. Parallelized stochastic gradient descent. In Advances in neural information processing systems. 2595--2603."},{"key":"e_1_3_2_1_50_1","unstructured":"Feng Zou. 2017. How to create ImageNet LMDB in Caffe. https:\/\/github.com\/intel\/caffe\/wiki\/How-to-create-Imagenet-LMDB.  Feng Zou. 2017. How to create ImageNet LMDB in Caffe. https:\/\/github.com\/intel\/caffe\/wiki\/How-to-create-Imagenet-LMDB."}],"event":{"name":"ICPP 2019: 48th International Conference on Parallel Processing","location":"Kyoto Japan","acronym":"ICPP 2019","sponsor":["University of Tsukuba University of Tsukuba"]},"container-title":["Proceedings of the 48th International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3337821.3337892","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3337821.3337892","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T00:25:41Z","timestamp":1750206341000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3337821.3337892"}},"subtitle":["Boosting End-to-End Deep Learning Workflows with Offloading Data Preprocessing Pipelines"],"short-title":[],"issued":{"date-parts":[[2019,8,5]]},"references-count":50,"alternative-id":["10.1145\/3337821.3337892","10.1145\/3337821"],"URL":"https:\/\/doi.org\/10.1145\/3337821.3337892","relation":{},"subject":[],"published":{"date-parts":[[2019,8,5]]},"assertion":[{"value":"2019-08-05","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}