{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T16:54:24Z","timestamp":1783788864092,"version":"3.55.0"},"publisher-location":"Cham","reference-count":23,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030787127","type":"print"},{"value":"9783030787134","type":"electronic"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-78713-4_7","type":"book-chapter","created":{"date-parts":[[2021,6,16]],"date-time":"2021-06-16T23:06:15Z","timestamp":1623884775000},"page":"118-136","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":23,"title":["Designing a ROCm-Aware MPI Library for AMD GPUs: Early Experiences"],"prefix":"10.1007","author":[{"given":"Kawthar","family":"Shafie Khorassani","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jahanzeb","family":"Hashmi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ching-Hsiang","family":"Chu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chen-Chun","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hari","family":"Subramoni","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dhabaleswar K.","family":"Panda","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,6,17]]},"reference":[{"key":"7_CR1","unstructured":"Bandwidth test for ROCm. https:\/\/github.com\/RadeonOpenCompute\/"},{"key":"7_CR2","unstructured":"Corona. https:\/\/hpc.llnl.gov\/hardware\/platforms\/corona"},{"key":"7_CR3","unstructured":"Frontier: ORNL\u2019s exascale supercomputer designed to deliver world-leading performance in 2021. https:\/\/www.olcf.ornl.gov\/frontier\/. Accessed 25 May 2021"},{"key":"7_CR4","unstructured":"Infiniband Verbs Performance Tests. https:\/\/github.com\/linux-rdma\/perftest"},{"key":"7_CR5","unstructured":"Radeon Open Compute (ROCm) Platform. https:\/\/rocmdocs.amd.com"},{"key":"7_CR6","unstructured":"RLLNL and HPE to partner with AMD on El Capitan, projected as world\u2019s fastest supercomputer. https:\/\/www.llnl.gov\/news\/llnl-and-hpe-partner-amd-el-capitan-projected-worlds-fastest-supercomputer. Accessed 25 May 2021"},{"key":"7_CR7","unstructured":"Unified Communication X. http:\/\/www.openucx.org\/. Accessed 25 May 2021"},{"key":"7_CR8","doi-asserted-by":"crossref","unstructured":"Cai, Z., et al.: Synthesizing optimal collective algorithms (2020)","DOI":"10.1145\/3437801.3441620"},{"key":"7_CR9","doi-asserted-by":"publisher","unstructured":"Chu, C.H., Khorassani, K.S., Zhou, Q., Subramoni, H., Panda, D.K.: Dynamic kernel fusion for bulk non-contiguous data transfer on gpu clusters. In: 2020 IEEE International Conference on Cluster Computing (CLUSTER), pp. 130\u2013141 (2020). https:\/\/doi.org\/10.1109\/CLUSTER49012.2020.00023","DOI":"10.1109\/CLUSTER49012.2020.00023"},{"issue":"1","key":"7_CR10","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1177\/1094342015593158","volume":"30","author":"J Dongarra","year":"2016","unstructured":"Dongarra, J., Heroux, M.A., Luszczek, P.: High-performance conjugate-gradient benchmark: a new metric for ranking high-performance computing systems. Int. J. High Perform. Comput. Appl. 30(1), 3\u201310 (2016)","journal-title":"Int. J. High Perform. Comput. Appl."},{"key":"7_CR11","unstructured":"Gabriel, E., et al.: Open MPI: goals, concept, and design of a next generation MPI implementation. In: Proceedings. 11th European PVM\/MPI Users\u2019 Group Meeting, Budapest, Hungary, pp. 97\u2013104, September 2004"},{"key":"7_CR12","doi-asserted-by":"publisher","unstructured":"Hashmi, J.M., Chu, C.H., Chakraborty, S., Bayatpour, M., Subramoni, H., Panda, D.K.: FALCON-X: zero-copy MPI derived datatype processing on modern CPU and GPU architectures. J. Parallel Distrib. Comput. 144, 1\u201313 (2020). https:\/\/doi.org\/10.1016\/j.jpdc.2020.05.008. http:\/\/www.sciencedirect.com\/science\/article\/pii\/S0743731520302872","DOI":"10.1016\/j.jpdc.2020.05.008"},{"key":"7_CR13","doi-asserted-by":"crossref","unstructured":"Khorassani, K.S., Chu, C.H., Subramoni, H., Panda, D.K.: Performance evaluation of MPI libraries on GPU-enabled OpenPOWER architectures: early experiences. In: International Workshop on OpenPOWER for HPC (IWOPH 19) at the 2019 ISC High Performance Conference (2018)","DOI":"10.1007\/978-3-030-34356-9_28"},{"key":"7_CR14","series-title":"Communications in Computer and Information Science","doi-asserted-by":"publisher","first-page":"121","DOI":"10.1007\/978-3-030-36592-9_11","volume-title":"Supercomputing","author":"E Kuznetsov","year":"2019","unstructured":"Kuznetsov, E., Stegailov, V.: Porting CUDA-based molecular dynamics algorithms to AMD ROCm platform using HIP framework: performance analysis. In: Voevodin, V., Sobolev, S. (eds.) RuSCDays 2019. CCIS, vol. 1129, pp. 121\u2013130. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-36592-9_11"},{"key":"7_CR15","doi-asserted-by":"publisher","unstructured":"Leiserson, C.E., et al.: There\u2019s plenty of room at the top: what will drive computer performance after Moore\u2019s law? Science 368(6495) (2020). https:\/\/doi.org\/10.1126\/science.aam9744. https:\/\/science.sciencemag.org\/content\/368\/6495\/eaam9744","DOI":"10.1126\/science.aam9744"},{"key":"7_CR16","doi-asserted-by":"publisher","unstructured":"Panda, D.K., Subramoni, H., Chu, C.H., Bayatpour, M.: The MVAPICH project: transforming research into high-performance MPI library for HPC community. J. Comput. Sci. 101208 (2020). https:\/\/doi.org\/10.1016\/j.jocs.2020.101208. http:\/\/www.sciencedirect.com\/science\/article\/pii\/S1877750320305093","DOI":"10.1016\/j.jocs.2020.101208"},{"key":"7_CR17","doi-asserted-by":"publisher","unstructured":"Potluri, S., Wang, H., Bureddy, D., Singh, A.K., Rosales, C., Panda, D.K.: Optimizing MPI communication on multi-GPU systems using CUDA inter-process communication. In: 2012 IEEE 26th International Parallel and Distributed Processing Symposium Workshops PhD Forum, pp. 1848\u20131857 (2012). https:\/\/doi.org\/10.1109\/IPDPSW.2012.228","DOI":"10.1109\/IPDPSW.2012.228"},{"key":"7_CR18","doi-asserted-by":"crossref","unstructured":"Potluri, S., Hamidouche, K., Venkatesh, A., Bureddy, D., Panda, D.K.: Efficient inter-node MPI communication using GPUDirect RDMA for InfiniBand clusters With NVIDIA GPUs. In: 2013 42nd International Conference on Parallel Processing (ICPP), pp. 80\u201389. IEEE (2013)","DOI":"10.1109\/ICPP.2013.17"},{"key":"7_CR19","doi-asserted-by":"crossref","unstructured":"Sharkawi, S.S., Chochia, G.A.: Communication protocol optimization for enhanced GPU performance. IBM J. Res. Dev. 64(3\/4), 9:1\u20139:9 (2020)","DOI":"10.1147\/JRD.2020.2967311"},{"key":"7_CR20","doi-asserted-by":"crossref","unstructured":"Shi, R., et al.: Designing efficient small message transfer mechanism for inter-node MPI communication on InfiniBand GPU clusters. In: 2014 21st International Conference on High Performance Computing (HiPC), pp. 1\u201310, December 2014","DOI":"10.1109\/HiPC.2014.7116873"},{"key":"7_CR21","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"334","DOI":"10.1007\/978-3-319-58667-0_18","volume-title":"High Performance Computing","author":"H Subramoni","year":"2017","unstructured":"Subramoni, H., Chakraborty, S., Panda, D.K.: Designing dynamic and adaptive MPI point-to-point communication protocols for efficient overlap of computation and communication. In: Kunkel, J.M., Yokota, R., Balaji, P., Keyes, D. (eds.) ISC 2017. LNCS, vol. 10266, pp. 334\u2013354. Springer, Cham (2017). https:\/\/doi.org\/10.1007\/978-3-319-58667-0_18"},{"key":"7_CR22","doi-asserted-by":"crossref","unstructured":"Tsai, Y.M., Cojean, T., Ribizel, T., Anzt, H.: Preparing ginkgo for AMD GPUS - a testimonial on porting CUDA code to HIP (2020)","DOI":"10.1007\/978-3-030-71593-9_9"},{"issue":"10","key":"7_CR23","doi-asserted-by":"publisher","first-page":"2595","DOI":"10.1109\/TPDS.2013.222","volume":"25","author":"H Wang","year":"2014","unstructured":"Wang, H., Potluri, S., Bureddy, D., Rosales, C., Panda, D.K.: GPU-aware MPI on RDMA-enabled clusters: design, implementation and evaluation. IEEE Trans. Parallel Distrib. Syst. 25(10), 2595\u20132605 (2014). https:\/\/doi.org\/10.1109\/TPDS.2013.222","journal-title":"IEEE Trans. Parallel Distrib. Syst."}],"container-title":["Lecture Notes in Computer Science","High Performance Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-78713-4_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,2]],"date-time":"2024-09-02T00:06:15Z","timestamp":1725235575000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-78713-4_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030787127","9783030787134"],"references-count":23,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-78713-4_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"17 June 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ISC High Performance","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on High Performance Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24 June 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 July 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"36","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"supercomputing2021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.isc-hpc.com\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Linklings","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"74","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"24","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"32% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4.28","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4.13","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"In the ISC High Performance Workshop, there were 49 submissions, out of which 35 were accepted.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}