{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:16:46Z","timestamp":1750220206344,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":41,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,8,29]],"date-time":"2022-08-29T00:00:00Z","timestamp":1661731200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["1818253, 1854828, 1931537, 2007991, 2018627"],"award-info":[{"award-number":["1818253, 1854828, 1931537, 2007991, 2018627"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,8,29]]},"DOI":"10.1145\/3547276.3548524","type":"proceedings-article","created":{"date-parts":[[2023,1,15]],"date-time":"2023-01-15T00:56:17Z","timestamp":1673744177000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Designing Hierarchical Multi-HCA Aware Allgather in MPI"],"prefix":"10.1145","author":[{"given":"Tu","family":"Tran","sequence":"first","affiliation":[{"name":"The Ohio State University, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Benjamin","family":"Michalowicz","sequence":"additional","affiliation":[{"name":"The Ohio State University, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bharath","family":"Ramesh","sequence":"additional","affiliation":[{"name":"The Ohio State University, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hari","family":"Subramoni","sequence":"additional","affiliation":[{"name":"The Ohio State University, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aamir","family":"Shafi","sequence":"additional","affiliation":[{"name":"The Ohio State University, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dhabaleswar K.","family":"Panda","sequence":"additional","affiliation":[{"name":"The Ohio State University, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,1,13]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"12th USENIX symposium on operating systems design and implementation (OSDI 16)","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi, Paul Barham, Jianmin Chen, Zhifeng Chen, Andy Davis, Jeffrey Dean, Matthieu Devin, Sanjay Ghemawat, Geoffrey Irving, Michael Isard, 2016. {TensorFlow}: A System for {Large-Scale} Machine Learning. In 12th USENIX symposium on operating systems design and implementation (OSDI 16). 265\u2013283."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-03770-2_9"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER.2018.00014"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.4851"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10586-021-03370-9"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER.2017.106"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/2020373.2020375"},{"key":"e_1_3_2_1_8_1","volume-title":"Retrieved","author":"Mukunoki Daichi","year":"2017","unstructured":"Daichi Mukunoki and Toshiyuki Imamura 2017. Implementation and Evaluation of 2.5D Matrix Multiplication on the K computer. Retrieved Mar 18, 2022 from https:\/\/prace-ri.eu\/wp-content\/uploads\/PRACE-at-SC17-Daichi-Mokunoki.pdf"},{"volume-title":"Enhancement of LiMIC-Based Collectives for Multi-core Clusters. Ph.\u00a0D. Dissertation","author":"Dhanraj Vijay","key":"e_1_3_2_1_9_1","unstructured":"Vijay Dhanraj. 2012. Enhancement of LiMIC-Based Collectives for Multi-core Clusters. Ph.\u00a0D. Dissertation. The Ohio State University."},{"key":"e_1_3_2_1_10_1","volume-title":"Retrieved","author":"Capitan El","year":"2022","unstructured":"El Capitan 2022. El Capitan. Retrieved Mar 18, 2022 from https:\/\/www.hpe.com\/us\/en\/newsroom\/press-release\/2020\/03\/hpe-and-amd-power-complex-scientific-discovery-in-worlds-fastest-supercomputer-for-us-department-of-energys-doe-national-nuclear-security-administration-nnsa.html"},{"key":"e_1_3_2_1_11_1","volume-title":"Retrieved","author":"Frontier","year":"2022","unstructured":"Frontier 2022. Frontier. Retrieved Mar 18, 2022 from https:\/\/www.olcf.ornl.gov\/frontier\/"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2018.00111"},{"key":"e_1_3_2_1_13_1","volume-title":"Retrieved","author":"HPC Advisory Council","year":"2022","unstructured":"HPC Advisory Council 2022. Thor. Retrieved Mar 18, 2022 from https:\/\/hpcadvisorycouncil.atlassian.net\/wiki\/spaces\/HPCWORKS\/pages\/7864401\/Thor"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2009.5160896"},{"key":"e_1_3_2_1_15_1","volume-title":"Retrieved","author":"Keras","year":"2022","unstructured":"Keras 2022. Keras Applications. Retrieved Mar 18, 2022 from https:\/\/keras.io\/api\/applications\/"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356176"},{"key":"e_1_3_2_1_17_1","volume-title":"SC\u201904: Proceedings of the 2004 ACM\/IEEE conference on Supercomputing. IEEE, 33\u201333","author":"Liu Jiuxing","year":"2004","unstructured":"Jiuxing Liu, Abhinav Vishnu, and Dhabaleswar\u00a0K Panda. 2004. Building multirail infiniband clusters: Mpi-level design and performance evaluation. In SC\u201904: Proceedings of the 2004 ACM\/IEEE conference on Supercomputing. IEEE, 33\u201333."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2011.29"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/11846802_17"},{"key":"e_1_3_2_1_20_1","volume-title":"MPI: A Message-Passing Interface Standard Version 4.0. https:\/\/www.mpi-forum.org\/docs\/mpi-4.0\/mpi40-report.pdf","author":"Interface Forum Message Passing","year":"2021","unstructured":"Message Passing Interface Forum. 2021. MPI: A Message-Passing Interface Standard Version 4.0. https:\/\/www.mpi-forum.org\/docs\/mpi-4.0\/mpi40-report.pdf"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCSE.2017.57"},{"key":"e_1_3_2_1_22_1","volume-title":"Osu network-based computing laboratory. URL: http:\/\/mvapich. cse. ohio-state. edu\/benchmarks 2","author":"Micro-Benchmarks OSU","year":"2018","unstructured":"OSU Micro-Benchmarks. 2018. Osu network-based computing laboratory. URL: http:\/\/mvapich. cse. ohio-state. edu\/benchmarks 2 (2018)."},{"key":"e_1_3_2_1_23_1","volume-title":"Retrieved","author":"Computing Laboratory Network-Based","year":"2022","unstructured":"Network-Based Computing Laboratory 2022. MVAPICH: MPI over InfiniBand, 10GigE\/iWARP and RoCE. Retrieved Mar 18, 2022 from http:\/\/mvapich.cse.ohio-state.edu\/"},{"key":"e_1_3_2_1_24_1","volume-title":"Retrieved","author":"NVIDIA","year":"2022","unstructured":"NVIDIA 2022. HPC-X. Retrieved Mar 18, 2022 from https:\/\/developer.nvidia.com\/networking\/hpc-x"},{"key":"e_1_3_2_1_25_1","volume-title":"MPI 2022","author":"Open","year":"2022","unstructured":"Open MPI 2022. Open MPI: Open Source High Performance Computing. Retrieved Mar 18, 2022 from https:\/\/www.open-mpi.org\/"},{"key":"e_1_3_2_1_26_1","volume-title":"Pytorch: An imperative style, high-performance deep learning library. Advances in neural information processing systems 32","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, 2019. Pytorch: An imperative style, high-performance deep learning library. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2008.09.002"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCS.2007.19"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390267"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2945397"},{"key":"e_1_3_2_1_31_1","unstructured":"Alexander Sergeev and Mike Del\u00a0Balso. 2018. Horovod: fast and easy distributed deep learning in TensorFlow. arXiv preprint arXiv:1802.05799(2018)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342006064482"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/11602569_19"},{"key":"e_1_3_2_1_34_1","first-page":"14","article-title":"MPI at Exascale","volume":"2","author":"Thakur Rajeev","year":"2010","unstructured":"Rajeev Thakur, Pavan Balaji, Darius Buntinas, David Goodell, William Gropp, Torsten Hoefler, Sameer Kumar, Ewing Lusk, and J\u00a0Larsson Tr\u00e4ff. 2010. MPI at Exascale. Procceedings of SciDAC 2(2010), 14\u201335.","journal-title":"Procceedings of SciDAC"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342005051521"},{"volume-title":"Retrieved","year":"2022","key":"e_1_3_2_1_36_1","unstructured":"ThetaGPU 2022. Theta\/ThetaGPU Machine Overview. Retrieved Mar 18, 2022 from https:\/\/www.alcf.anl.gov\/support-center\/theta\/theta-thetagpu-overview"},{"volume-title":"NOVEMBER 2021","year":"2022","key":"e_1_3_2_1_37_1","unstructured":"Top500 2022. NOVEMBER 2021. Retrieved Mar 18, 2022 from https:\/\/www.top500.org\/lists\/top500\/2021\/11\/"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER49012.2020.00037"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2017.05.009"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2014.115"},{"key":"e_1_3_2_1_41_1","volume-title":"Proceedings of the 48th International Conference on Parallel Processing: Workshops. 1\u201310","author":"Zhou Huan","year":"2019","unstructured":"Huan Zhou, Jos\u00e9 Gracia, and Ralf Schneider. 2019. MPI collectives for multi-core clusters: Optimized performance of the hybrid MPI+ MPI parallel codes. In Proceedings of the 48th International Conference on Parallel Processing: Workshops. 1\u201310."}],"event":{"name":"ICPP '22: 51st International Conference on Parallel Processing","acronym":"ICPP '22","location":"Bordeaux France"},"container-title":["Workshop Proceedings of the 51st International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3547276.3548524","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3547276.3548524","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3547276.3548524","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:56Z","timestamp":1750186976000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3547276.3548524"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,29]]},"references-count":41,"alternative-id":["10.1145\/3547276.3548524","10.1145\/3547276"],"URL":"https:\/\/doi.org\/10.1145\/3547276.3548524","relation":{},"subject":[],"published":{"date-parts":[[2022,8,29]]},"assertion":[{"value":"2023-01-13","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}