{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T01:43:27Z","timestamp":1787017407233,"version":"build-2736575974"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2017,11,12]],"date-time":"2017-11-12T00:00:00Z","timestamp":1510444800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2017,11,12]]},"DOI":"10.1145\/3146347.3146356","type":"proceedings-article","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T08:31:37Z","timestamp":1509438697000},"page":"1-8","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":62,"title":["An In-depth Performance Characterization of CPU- and GPU-based DNN Training on Modern Architectures"],"prefix":"10.1145","author":[{"given":"Ammar Ahmad","family":"Awan","sequence":"first","affiliation":[{"name":"Dept. of Computer Science and Engg., The Ohio State University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hari","family":"Subramoni","sequence":"additional","affiliation":[{"name":"Dept. of Computer Science and Engg., The Ohio State University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dhabaleswar K.","family":"Panda","sequence":"additional","affiliation":[{"name":"Dept. of Computer Science and Engg., The Ohio State University"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2017,11,12]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"http:\/\/caffe.berkeleyvision.org\/. (2017). [Online","author":"Website Caffe","year":"2017","unstructured":"2017. Caffe Website . http:\/\/caffe.berkeleyvision.org\/. (2017). [Online ; accessed Sep- 2017 ]. 2017. Caffe Website. http:\/\/caffe.berkeleyvision.org\/. (2017). [Online; accessed Sep-2017]."},{"key":"e_1_3_2_1_2_1","unstructured":"2017. Caffe2 Framework. https:\/\/caffe2.ai. (2017).  2017. Caffe2 Framework. https:\/\/caffe2.ai. (2017)."},{"key":"e_1_3_2_1_3_1","unstructured":"2017. cuBLAS Library. https:\/\/developer.nvidia.com\/cublas. (2017).  2017. cuBLAS Library. https:\/\/developer.nvidia.com\/cublas. (2017)."},{"key":"e_1_3_2_1_4_1","unstructured":"2017. cuDNN Library. https:\/\/developer.nvidia.com\/cudnn. (2017).  2017. cuDNN Library. https:\/\/developer.nvidia.com\/cudnn. (2017)."},{"key":"e_1_3_2_1_5_1","volume-title":"https:\/\/github.com\/intelcaffe. (2017). [Online","author":"Caffe Intel","year":"2017","unstructured":"2017. Intel Caffe . https:\/\/github.com\/intelcaffe. (2017). [Online ; accessed Sep- 2017 ]. 2017. Intel Caffe. https:\/\/github.com\/intelcaffe. (2017). [Online; accessed Sep-2017]."},{"key":"e_1_3_2_1_6_1","volume-title":"https:\/\/github.com\/NVIDIA\/caffe. (2017). [Online","author":"Caffe Nvidia","year":"2017","unstructured":"2017. Nvidia Caffe . https:\/\/github.com\/NVIDIA\/caffe. (2017). [Online ; accessed Sep- 2017 ]. 2017. Nvidia Caffe. https:\/\/github.com\/NVIDIA\/caffe. (2017). [Online; accessed Sep-2017]."},{"key":"e_1_3_2_1_7_1","unstructured":"2017. Xeon Phi Memory Latency. https:\/\/sites.utexas.edu\/jdm4372\/2016\/12\/06\/memory-latency-on-the-intel-xeon-phi-x200-knights-landing-processor\/. (2017).  2017. Xeon Phi Memory Latency. https:\/\/sites.utexas.edu\/jdm4372\/2016\/12\/06\/memory-latency-on-the-intel-xeon-phi-x200-knights-landing-processor\/. (2017)."},{"key":"e_1_3_2_1_8_1","unstructured":"M. Abadi A. Agarwal P. Barham E. Brevdo Z. Chen C. Citro G. S. Corrado A. Davis J. Dean M. Devin etal 2016. TensorFlow: Large-Scale Machine Learning on Heterogeneous Systems 2015. arXiv preprint arXiv:1603.04467 (2016).  M. Abadi A. Agarwal P. Barham E. Brevdo Z. Chen C. Citro G. S. Corrado A. Davis J. Dean M. Devin et al. 2016. TensorFlow: Large-Scale Machine Learning on Heterogeneous Systems 2015. arXiv preprint arXiv:1603.04467 (2016)."},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of the 22nd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (To be presented) (PPoPP '17)","author":"Awan A. A.","unstructured":"A. A. Awan , K. Hamidouche , J. M. Hashmi , and D. K. Panda . 2017. S-Caffe: Co-designing MPI Runtimes and Caffe for Scalable Deep Learning on Modern GPU Clusters . In Proceedings of the 22nd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (To be presented) (PPoPP '17) . ACM, New York, NY, USA. A. A. Awan, K. Hamidouche, J. M. Hashmi, and D. K. Panda. 2017. S-Caffe: Co-designing MPI Runtimes and Caffe for Scalable Deep Learning on Modern GPU Clusters. In Proceedings of the 22nd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (To be presented) (PPoPP '17). ACM, New York, NY, USA."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_11_1","volume-title":"Inside Pascal: NVIDIA's Newest Computing Platform. https:\/\/devblogs.nvidia.com\/parallelforall\/insidepascal\/.","author":"Harris Mark","year":"2017","unstructured":"Mark Harris . 2017 . Inside Pascal: NVIDIA's Newest Computing Platform. https:\/\/devblogs.nvidia.com\/parallelforall\/insidepascal\/. (2017). [Online; accessed Aug-2017]. Mark Harris. 2017. Inside Pascal: NVIDIA's Newest Computing Platform. https:\/\/devblogs.nvidia.com\/parallelforall\/insidepascal\/. (2017). [Online; accessed Aug-2017]."},{"key":"e_1_3_2_1_12_1","volume":"201","author":"He K.","unstructured":"K. He , X. Zhang , S. Ren , and J. Sun. 201 5. Deep Residual Learning for Image Recognition. ArXiv e-prints (Dec. 2015). arXiv:cs.CV\/1512.03385 K. He, X. Zhang, S. Ren, and J. Sun. 2015. Deep Residual Learning for Image Recognition. ArXiv e-prints (Dec. 2015). arXiv:cs.CV\/1512.03385","journal-title":"J. Sun."},{"key":"e_1_3_2_1_13_1","unstructured":"Intel. 2017. Machine Learning Scaling Library. https:\/\/github.com\/01org\/MLSL. (2017).  Intel. 2017. Machine Learning Scaling Library. https:\/\/github.com\/01org\/MLSL. (2017)."},{"key":"e_1_3_2_1_14_1","unstructured":"Intel 2017. MKL-DNN for Scalable Deep Learning. https:\/\/software.intel.com\/en-us\/articles\/introducing-dnn-primitives-in-intelr-mkl. (2017).  Intel 2017. MKL-DNN for Scalable Deep Learning. https:\/\/software.intel.com\/en-us\/articles\/introducing-dnn-primitives-in-intelr-mkl. (2017)."},{"key":"e_1_3_2_1_15_1","volume-title":"Caffe: Convolutional Architecture for Fast Feature Embedding. arXiv preprint arXiv:1408.5093","author":"Jia Y.","year":"2014","unstructured":"Y. Jia , E. Shelhamer , J. Donahue , S. Karayev , J. Long , R. Girshick , S. Guadarrama , and T. Darrell . 2014 . Caffe: Convolutional Architecture for Fast Feature Embedding. arXiv preprint arXiv:1408.5093 (2014). Y. Jia, E. Shelhamer, J. Donahue, S. Karayev, J. Long, R. Girshick, S. Guadarrama, and T. Darrell. 2014. Caffe: Convolutional Architecture for Fast Feature Embedding. arXiv preprint arXiv:1408.5093 (2014)."},{"key":"e_1_3_2_1_16_1","volume-title":"One Weird Trick for Parallelizing Convolutional Neural Networks. CoRR abs\/1404.5997","author":"Krizhevsky A.","year":"2014","unstructured":"A. Krizhevsky . 2014. One Weird Trick for Parallelizing Convolutional Neural Networks. CoRR abs\/1404.5997 ( 2014 ). A. Krizhevsky. 2014. One Weird Trick for Parallelizing Convolutional Neural Networks. CoRR abs\/1404.5997 (2014)."},{"key":"e_1_3_2_1_17_1","unstructured":"A. Krizhevsky and G. Hinton. 2009. Learning Multiple Layers of Features from Tiny Images. http:\/\/www.cs.toronto.edu\/~kriz\/learning-features-2009-TR.pdf. (2009).  A. Krizhevsky and G. Hinton. 2009. Learning Multiple Layers of Features from Tiny Images. http:\/\/www.cs.toronto.edu\/~kriz\/learning-features-2009-TR.pdf. (2009)."},{"key":"e_1_3_2_1_18_1","unstructured":"A. Krizhevsky I. Sutskever and G. E. Hinton. 2012. ImageNet Classification with Deep Convolutional Neural Networks. In Advances in Neural Information Processing Systems 25. 1097-- 1105.  A. Krizhevsky I. Sutskever and G. E. Hinton. 2012. ImageNet Classification with Deep Convolutional Neural Networks. In Advances in Neural Information Processing Systems 25. 1097-- 1105."},{"key":"e_1_3_2_1_19_1","volume-title":"http:\/\/www.cntk.ai\/. (2017). [Online","author":"CNTK.","year":"2017","unstructured":"Microsoft. 2017. CNTK. http:\/\/www.cntk.ai\/. (2017). [Online ; accessed April- 2017 ]. Microsoft. 2017. CNTK. http:\/\/www.cntk.ai\/. (2017). [Online; accessed April-2017]."},{"key":"e_1_3_2_1_20_1","volume-title":"Omni-Path, Ethernet\/iWARP, and RoCE.","author":"MVAPICH","year":"2017","unstructured":"MVAPICH : MPI over InfiniBand , Omni-Path, Ethernet\/iWARP, and RoCE. 2017 . https:\/\/mvapich.cse.ohio-state.edu\/. (2017). MVAPICH: MPI over InfiniBand, Omni-Path, Ethernet\/iWARP, and RoCE. 2017. https:\/\/mvapich.cse.ohio-state.edu\/. (2017)."},{"key":"e_1_3_2_1_21_1","volume-title":"d.]. Programming Guide. ([n. d.]). [Online","author":"Nvidia CUDA","year":"2017","unstructured":"CUDA Nvidia . [n. d.]. Programming Guide. ([n. d.]). [Online ; accessed April- 2017 ]. CUDA Nvidia. [n. d.]. Programming Guide. ([n. d.]). [Online; accessed April-2017]."},{"key":"e_1_3_2_1_22_1","unstructured":"D. K. Panda A. A. Awan and H. Subramoni. 2017. High Performance Distributed Deep Learning for Dummies. Tutorial presented at Hot Interconnects 17 (2017). http:\/\/www.hoti.org\/hoti25\/tutorials\/  D. K. Panda A. A. Awan and H. Subramoni. 2017. High Performance Distributed Deep Learning for Dummies. Tutorial presented at Hot Interconnects 17 (2017). http:\/\/www.hoti.org\/hoti25\/tutorials\/"},{"key":"e_1_3_2_1_23_1","unstructured":"S. Shi. 2017. Benchmarking State-of-the-Art Deep Learning Software Tools. http:\/\/dlbench.comp.hkbu.edu.hk\/. (2017). [Online; accessed Aug-2017].  S. Shi. 2017. Benchmarking State-of-the-Art Deep Learning Software Tools. http:\/\/dlbench.comp.hkbu.edu.hk\/. (2017). [Online; accessed Aug-2017]."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/HOTCHIPS.2015.7477467"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_1_26_1","unstructured":"HiDL Team. 2017. High Performance Deep Learning Project. http:\/\/hidl.cse.ohio-state.edu. (2017).  HiDL Team. 2017. High Performance Deep Learning Project. http:\/\/hidl.cse.ohio-state.edu. (2017)."}],"event":{"name":"SC '17: The International Conference for High Performance Computing, Networking, Storage and Analysis","location":"Denver CO USA","acronym":"SC '17","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","IEEE CS"]},"container-title":["Proceedings of the Machine Learning on HPC Environments"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3146347.3146356","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3146347.3146356","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T22:13:33Z","timestamp":1750198413000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3146347.3146356"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,11,12]]},"references-count":26,"alternative-id":["10.1145\/3146347.3146356","10.1145\/3146347"],"URL":"https:\/\/doi.org\/10.1145\/3146347.3146356","relation":{},"subject":[],"published":{"date-parts":[[2017,11,12]]},"assertion":[{"value":"2017-11-12","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}