{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T15:37:44Z","timestamp":1772725064547,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2016,9,25]],"date-time":"2016-09-25T00:00:00Z","timestamp":1474761600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2016,9,25]]},"DOI":"10.1145\/2966884.2966912","type":"proceedings-article","created":{"date-parts":[[2016,10,20]],"date-time":"2016-10-20T15:31:56Z","timestamp":1476977516000},"page":"15-22","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":39,"title":["Efficient Large Message Broadcast using NCCL and CUDA-Aware MPI for Deep Learning"],"prefix":"10.1145","author":[{"given":"A. A.","family":"Awan","sequence":"first","affiliation":[{"name":"Dept of Computer Science and Engineering, The Ohio State University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"K.","family":"Hamidouche","sequence":"additional","affiliation":[{"name":"Dept of Computer Science and Engineering, The Ohio State University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"A.","family":"Venkatesh","sequence":"additional","affiliation":[{"name":"Dept of Computer Science and Engineering, The Ohio State University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"D. K.","family":"Panda","sequence":"additional","affiliation":[{"name":"Dept of Computer Science and Engineering, The Ohio State University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2016,9,25]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"KESCH\n  : Cray CS-Storm System (CSCS). http:\/\/www.cscs.ch\/computers\/kesch_escha\/index.html.  KESCH: Cray CS-Storm System (CSCS). http:\/\/www.cscs.ch\/computers\/kesch_escha\/index.html."},{"key":"e_1_3_2_1_2_1","volume-title":"http:\/\/www.cntk.ai\/","author":"CNTK.","year":"2015","unstructured":"CNTK. http:\/\/www.cntk.ai\/ , 2015 . CNTK. http:\/\/www.cntk.ai\/, 2015."},{"key":"e_1_3_2_1_3_1","volume-title":"Deep Scalable Sparse Tensor Network Engine. https:\/\/github.com\/amznlabs\/amazon-dsstne","year":"2016","unstructured":"Amazon. Deep Scalable Sparse Tensor Network Engine. https:\/\/github.com\/amznlabs\/amazon-dsstne , 2016 . Amazon. Deep Scalable Sparse Tensor Network Engine. https:\/\/github.com\/amznlabs\/amazon-dsstne, 2016."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/SHPCC.1994.296665"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33518-1_16"},{"key":"e_1_3_2_1_6_1","unstructured":"Cray. Cray Message Passing Toolkit. http:\/\/docs.cray.com\/books\/S-9407-1305\/S-9407-1305.pdf 2015.  Cray. Cray Message Passing Toolkit. http:\/\/docs.cray.com\/books\/S-9407-1305\/S-9407-1305.pdf 2015."},{"key":"e_1_3_2_1_7_1","unstructured":"Cray. CS-Storm Cluster Supercomputer System. http:\/\/www.cray.com\/products\/computing\/cs-series\/cs-storm 2016.  Cray. CS-Storm Cluster Supercomputer System. http:\/\/www.cray.com\/products\/computing\/cs-series\/cs-storm 2016."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ExaMPI.2014.5"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33518-1_18"},{"key":"e_1_3_2_1_11_1","volume-title":"FireCaffe: Near-Linear Acceleration of Deep Neural Network Training on Compute Clusters. arXiv preprint arXiv:1511.00175","author":"Iandola F. N.","year":"2015","unstructured":"F. N. Iandola , K. Ashraf , M. W. Moskewicz , and K. Keutzer . FireCaffe: Near-Linear Acceleration of Deep Neural Network Training on Compute Clusters. arXiv preprint arXiv:1511.00175 , 2015 . F. N. Iandola, K. Ashraf, M. W. Moskewicz, and K. Keutzer. FireCaffe: Near-Linear Acceleration of Deep Neural Network Training on Compute Clusters. arXiv preprint arXiv:1511.00175, 2015."},{"key":"e_1_3_2_1_12_1","unstructured":"Intel. Big Data Meets High Performance Computing: High Performance Data Analytics. http:\/\/www.intel.com\/content\/dam\/www\/public\/us\/en\/documents\/white-papers\/big-data-meets-high-performance-computing-white-paper.pdf.  Intel. Big Data Meets High Performance Computing: High Performance Data Analytics. http:\/\/www.intel.com\/content\/dam\/www\/public\/us\/en\/documents\/white-papers\/big-data-meets-high-performance-computing-white-paper.pdf."},{"key":"e_1_3_2_1_13_1","unstructured":"Intel Weighs In on NSCI. Intel Weighs In on NSCI. http:\/\/www.hpcwire.com\/2016\/05\/05\/intel-weighs-nsci\/.  Intel Weighs In on NSCI. Intel Weighs In on NSCI. http:\/\/www.hpcwire.com\/2016\/05\/05\/intel-weighs-nsci\/."},{"key":"e_1_3_2_1_14_1","volume-title":"Caffe: Convolutional Architecture for Fast Feature Embedding. arXiv preprint arXiv:1408.5093","author":"Jia Y.","year":"2014","unstructured":"Y. Jia , E. Shelhamer , J. Donahue , S. Karayev , J. Long , R. Girshick , S. Guadarrama , and T. Darrell . Caffe: Convolutional Architecture for Fast Feature Embedding. arXiv preprint arXiv:1408.5093 , 2014 . Y. Jia, E. Shelhamer, J. Donahue, S. Karayev, J. Long, R. Girshick, S. Guadarrama, and T. Darrell. Caffe: Convolutional Architecture for Fast Feature Embedding. arXiv preprint arXiv:1408.5093, 2014."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/HOTI.2013.26"},{"key":"e_1_3_2_1_16_1","first-page":"1097","volume-title":"Advances in Neural Information Processing Systems","author":"Krizhevsky A.","year":"2012","unstructured":"A. Krizhevsky , I. Sutskever , and G. E. Hinton . ImageNet Classification with Deep Convolutional Neural Networks . In Advances in Neural Information Processing Systems , pages 1097 -- 1105 , 2012 . A. Krizhevsky, I. Sutskever, and G. E. Hinton. ImageNet Classification with Deep Convolutional Neural Networks. In Advances in Neural Information Processing Systems, pages 1097--1105, 2012."},{"key":"e_1_3_2_1_17_1","first-page":"10","volume-title":"Parallel and Distributed Processing Symposium, 2004. Proceedings. 18th International","author":"Liu J.","unstructured":"J. Liu , A. R. Mamidala , and D. K. Panda . Fast and Scalable MPI-level Broadcast using InfiniBand's Hardware Multicast Support . In Parallel and Distributed Processing Symposium, 2004. Proceedings. 18th International , pages 10 --, April 2004. J. Liu, A. R. Mamidala, and D. K. Panda. Fast and Scalable MPI-level Broadcast using InfiniBand's Hardware Multicast Support. In Parallel and Distributed Processing Symposium, 2004. Proceedings. 18th International, pages 10--, April 2004."},{"key":"e_1_3_2_1_18_1","unstructured":"MVAPICH2: MPI over InfiniBand 10GigE\/iWARP and RoCE. https:\/\/mvapich.cse.ohio-state.edu\/.  MVAPICH2: MPI over InfiniBand 10GigE\/iWARP and RoCE. https:\/\/mvapich.cse.ohio-state.edu\/."},{"key":"e_1_3_2_1_19_1","volume-title":"Fast Multi-GPU Collectives with NCCL (Nickel). https:\/\/devblogs.nvidia.com\/parallelforall\/fast-multi-gpu-collectives-nccl\/","author":"Nathan Luehr","year":"2016","unstructured":"Nathan Luehr (NVIDIA). Fast Multi-GPU Collectives with NCCL (Nickel). https:\/\/devblogs.nvidia.com\/parallelforall\/fast-multi-gpu-collectives-nccl\/ , 2016 . Nathan Luehr (NVIDIA). Fast Multi-GPU Collectives with NCCL (Nickel). https:\/\/devblogs.nvidia.com\/parallelforall\/fast-multi-gpu-collectives-nccl\/, 2016."},{"key":"e_1_3_2_1_20_1","volume-title":"http:\/\/mvapich.cse.ohio-state.edu\/benchmarks\/","author":"Computing Laboratory Network Based","year":"2015","unstructured":"Network Based Computing Laboratory . OSU Micro-Benchmarks . http:\/\/mvapich.cse.ohio-state.edu\/benchmarks\/ , 2015 . Network Based Computing Laboratory. OSU Micro-Benchmarks. http:\/\/mvapich.cse.ohio-state.edu\/benchmarks\/, 2015."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.5555\/1855591.1855609"},{"key":"e_1_3_2_1_22_1","unstructured":"Nvidia. DGX-1: Deep Learning Supercomputer. http:\/\/www.nvidia.com\/object\/deep-learning-system.html 2016.  Nvidia. DGX-1: Deep Learning Supercomputer. http:\/\/www.nvidia.com\/object\/deep-learning-system.html 2016."},{"key":"e_1_3_2_1_23_1","volume-title":"https:\/\/github.com\/NVIDIA\/nccl","author":"Library NCCL","year":"2016","unstructured":"Nvidia. NCCL Library . https:\/\/github.com\/NVIDIA\/nccl , 2016 . Nvidia. NCCL Library. https:\/\/github.com\/NVIDIA\/nccl, 2016."},{"key":"e_1_3_2_1_24_1","volume-title":"Very Deep Convolutional Networks for Large-Scale Image Recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan K.","year":"2014","unstructured":"K. Simonyan and A. Zisserman . Very Deep Convolutional Networks for Large-Scale Image Recognition. arXiv preprint arXiv:1409.1556 , 2014 . K. Simonyan and A. Zisserman. Very Deep Convolutional Networks for Large-Scale Image Recognition. arXiv preprint arXiv:1409.1556, 2014."},{"key":"e_1_3_2_1_25_1","unstructured":"The Open MPI Development Team. Open MPI: Open Source High Performance Computing. http:\/\/www.open-mpi.org.  The Open MPI Development Team. Open MPI: Open Source High Performance Computing. http:\/\/www.open-mpi.org."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPPW.2015.20"}],"event":{"name":"EuroMPI 2016: The 23rd European MPI Users' Group Meeting","location":"Edinburgh United Kingdom","acronym":"EuroMPI 2016","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the 23rd European MPI Users' Group Meeting"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2966884.2966912","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2966884.2966912","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:55:53Z","timestamp":1750222553000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2966884.2966912"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,9,25]]},"references-count":26,"alternative-id":["10.1145\/2966884.2966912","10.1145\/2966884"],"URL":"https:\/\/doi.org\/10.1145\/2966884.2966912","relation":{},"subject":[],"published":{"date-parts":[[2016,9,25]]},"assertion":[{"value":"2016-09-25","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}