{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,10]],"date-time":"2026-01-10T08:10:36Z","timestamp":1768032636869,"version":"3.49.0"},"publisher-location":"Cham","reference-count":28,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030343552","type":"print"},{"value":"9783030343569","type":"electronic"}],"license":[{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-030-34356-9_28","type":"book-chapter","created":{"date-parts":[[2019,12,2]],"date-time":"2019-12-02T18:37:03Z","timestamp":1575311823000},"page":"361-378","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":15,"title":["Performance Evaluation of MPI Libraries on GPU-Enabled OpenPOWER Architectures: Early Experiences"],"prefix":"10.1007","author":[{"given":"Kawthar Shafie","family":"Khorassani","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ching-Hsiang","family":"Chu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hari","family":"Subramoni","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dhabaleswar K.","family":"Panda","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,12,3]]},"reference":[{"key":"28_CR1","unstructured":"IBM Spectrum MPI version 10.3. \nhttps:\/\/www.ibm.com"},{"key":"28_CR2","unstructured":"Infiniband Verbs Performance Tests. \nhttps:\/\/github.com\/linux-rdma\/perftest\n\n. Accessed 26 Oct 2019"},{"key":"28_CR3","unstructured":"MVAPICH: MPI over InfiniBand, Omni-Path, Ethernet\/iWARP, and RoCE. \nhttp:\/\/mvapich.cse.ohio-state.edu\/features\/"},{"key":"28_CR4","unstructured":"Open MPI: Open Source High Performance Computing. \nhttps:\/\/www.open-mpi.org"},{"key":"28_CR5","unstructured":"TOP 500 Supercomputer Sites. \nhttp:\/\/www.top500.org"},{"key":"28_CR6","unstructured":"Unified Communication X. \nhttp:\/\/www.openucx.org\/\n\n. Accessed 26 Oct 2019"},{"key":"28_CR7","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"173","DOI":"10.1007\/978-3-319-46079-6_13","volume-title":"High Performance Computing","author":"M Ashworth","year":"2016","unstructured":"Ashworth, M., Meng, J., Novakovic, V., Siso, S.: Early application performance at the hartree centre with the OpenPOWER architecture. In: Taufer, M., Mohr, B., Kunkel, J.M. (eds.) ISC High Performance 2016. LNCS, vol. 9945, pp. 173\u2013187. Springer, Cham (2016). \nhttps:\/\/doi.org\/10.1007\/978-3-319-46079-6_13"},{"key":"28_CR8","doi-asserted-by":"crossref","unstructured":"Awan, A.A., B\u00e9dorf, J., Chu, C.H., Subramoni, H., Panda, D.K.: Scalable distributed DNN training using TensorFlow and CUDA-aware MPI: characterization, designs, and performance evaluation. In: The 19th Annual IEEE\/ACM International Symposium in Cluster, Cloud, and Grid Computing (CCGRID 2019) (2019)","DOI":"10.1109\/CCGRID.2019.00064"},{"key":"28_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"110","DOI":"10.1007\/978-3-642-33518-1_16","volume-title":"Recent Advances in the Message Passing Interface","author":"D Bureddy","year":"2012","unstructured":"Bureddy, D., Wang, H., Venkatesh, A., Potluri, S., Panda, D.K.: OMB-GPU: a micro-benchmark suite for evaluating MPI libraries on GPU clusters. In: Tr\u00e4ff, J.L., Benkner, S., Dongarra, J.J. (eds.) EuroMPI 2012. LNCS, vol. 7490, pp. 110\u2013120. Springer, Heidelberg (2012). \nhttps:\/\/doi.org\/10.1007\/978-3-642-33518-1_16"},{"key":"28_CR10","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"448","DOI":"10.1007\/978-3-030-02465-9_32","volume-title":"High Performance Computing","author":"C Pearson","year":"2018","unstructured":"Pearson, C., Chung, I.-H., Sura, Z., Hwu, W.-M., Xiong, J.: NUMA-aware data-transfer measurements for power\/NVLink multi-GPU systems. In: Yokota, R., Weiland, M., Shalf, J., Alam, S. (eds.) ISC High Performance 2018. LNCS, vol. 11203, pp. 448\u2013454. Springer, Cham (2018). \nhttps:\/\/doi.org\/10.1007\/978-3-030-02465-9_32"},{"key":"28_CR11","doi-asserted-by":"crossref","unstructured":"Chu, C.H., Hamidouche, K., Venkatesh, A., Banerjee, D.S., Subramoni, H., Panda, D.K.: Exploiting maximal overlap for non-contiguous data movement processing on modern GPU-enabled systems. In: 2016 IEEE International Parallel and Distributed Processing Symposium (IPDPS), pp. 983\u2013992, May 2016","DOI":"10.1109\/IPDPS.2016.99"},{"key":"28_CR12","doi-asserted-by":"crossref","unstructured":"Chu, C.H., et al.: Efficient and scalable multi-source streaming broadcast on GPU clusters for deep learning. In: 46th International Conference on Parallel Processing (ICPP-2017), August 2017","DOI":"10.1109\/ICPP.2017.25"},{"issue":"2","key":"28_CR13","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1109\/MM.2017.37","volume":"37","author":"D Foley","year":"2017","unstructured":"Foley, D., Danskin, J.: Ultra-performance pascal GPU and NVLink interconnect. IEEE Micro 37(2), 7\u201317 (2017). \nhttps:\/\/doi.org\/10.1109\/MM.2017.37","journal-title":"IEEE Micro"},{"key":"28_CR14","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"97","DOI":"10.1007\/978-3-540-30218-6_19","volume-title":"Recent Advances in Parallel Virtual Machine and Message Passing Interface","author":"E Gabriel","year":"2004","unstructured":"Gabriel, E., et al.: Open MPI: goals, concept, and design of a next generation MPI implementation. In: Kranzlm\u00fcller, D., Kacsuk, P., Dongarra, J. (eds.) EuroPVM\/MPI 2004. LNCS, vol. 3241, pp. 97\u2013104. Springer, Heidelberg (2004). \nhttps:\/\/doi.org\/10.1007\/978-3-540-30218-6_19"},{"key":"28_CR15","unstructured":"McCalpin, J.D.: STREAM: sustainable memory bandwidth in high performance computers (2019). \nhttps:\/\/www.cs.virginia.edu\/stream\/\n\n. Accessed 26 Oct 2019"},{"key":"28_CR16","unstructured":"Li, A., et al.: Evaluating modern GPU interconnect: PCIe, NVLink, NV-SLI, NVSwitch and GPUDirect. CoRR abs\/1903.04611 (2019). \nhttp:\/\/arxiv.org\/abs\/1903.04611"},{"key":"28_CR17","doi-asserted-by":"publisher","unstructured":"Luo, X., Wu, W., Bosilca, G., Patinyasakdikul, T., Wang, L., Dongarra, J.: ADAPT: an event-based adaptive collective communication framework. In: Proceedings of the 27th International Symposium on High-Performance Parallel and Distributed Computing, HPDC 2018, pp. 118\u2013130. ACM, New York (2018). \nhttps:\/\/doi.org\/10.1145\/3208040.3208054","DOI":"10.1145\/3208040.3208054"},{"key":"28_CR18","doi-asserted-by":"publisher","unstructured":"Mojumder, S.A., et al.: Profiling DNN workloads on a volta-based DGX-1 system. In: 2018 IEEE International Symposium on Workload Characterization (IISWC), pp. 122\u2013133, September 2018. \nhttps:\/\/doi.org\/10.1109\/IISWC.2018.8573521","DOI":"10.1109\/IISWC.2018.8573521"},{"key":"28_CR19","doi-asserted-by":"publisher","unstructured":"Moreno, R., Arias, E., Navarro, A., Tapiador, F.J.: How good is the OpenPOWER architecture for high-performance CPU-oriented weather forecasting applications? J. Supercomput., April 2019. \nhttps:\/\/doi.org\/10.1007\/s11227-019-02844-3","DOI":"10.1007\/s11227-019-02844-3"},{"key":"28_CR20","unstructured":"NVIDIA: NVIDIA GPUDirect. \nhttps:\/\/developer.nvidia.com\/gpudirect\n\n. Accessed 26 Oct 2019"},{"key":"28_CR21","unstructured":"NVIDIA: NVIDIA Tesla V100 GPU Architecture (2019). \nhttps:\/\/images.nvidia.com\/content\/volta-architecture\/pdf\/volta-architecture-whitepaper.pdf\n\n. Accessed 26 Oct 2019"},{"key":"28_CR22","first-page":"617","volume":"42","author":"GF Pfister","year":"2001","unstructured":"Pfister, G.F.: An introduction to the infiniband architecture. High Perform. Mass Storage Parallel I\/O 42, 617\u2013632 (2001)","journal-title":"High Perform. Mass Storage Parallel I\/O"},{"key":"28_CR23","doi-asserted-by":"crossref","unstructured":"Potluri, S., Hamidouche, K., Venkatesh, A., Bureddy, D., Panda, D.K.: Efficient inter-node MPI communication using GPUDirect RDMA for InfiniBand clusters with NVIDIA GPUs. In: 2013 42nd International Conference on Parallel Processing (ICPP), pp. 80\u201389. IEEE (2013)","DOI":"10.1109\/ICPP.2013.17"},{"key":"28_CR24","doi-asserted-by":"crossref","unstructured":"Shi, R., et al.: Designing efficient small message transfer mechanism for inter-node MPI communication on InfiniBand GPU clusters. In: 2014 21st International Conference on High Performance Computing (HiPC), pp. 1\u201310, December 2014","DOI":"10.1109\/HiPC.2014.7116873"},{"key":"28_CR25","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"188","DOI":"10.1007\/978-3-319-46079-6_14","volume-title":"High Performance Computing","author":"JE Stone","year":"2016","unstructured":"Stone, J.E., Hynninen, A.-P., Phillips, J.C., Schulten, K.: Early experiences porting the NAMD and VMD molecular simulation and analysis software to GPU-accelerated OpenPOWER platforms. In: Taufer, M., Mohr, B., Kunkel, J.M. (eds.) ISC High Performance 2016. LNCS, vol. 9945, pp. 188\u2013206. Springer, Cham (2016). \nhttps:\/\/doi.org\/10.1007\/978-3-319-46079-6_14"},{"key":"28_CR26","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1007\/978-3-319-72971-8_1","volume-title":"High Performance Computing Systems. Performance Modeling, Benchmarking, and Simulation","author":"NR Tallent","year":"2018","unstructured":"Tallent, N.R., Gawande, N.A., Siegel, C., Vishnu, A., Hoisie, A.: Evaluating on-node GPU interconnects for\u00a0deep learning workloads. In: Jarvis, S., Wright, S., Hammond, S. (eds.) PMBS 2017. LNCS, vol. 10724, pp. 3\u201321. Springer, Cham (2018). \nhttps:\/\/doi.org\/10.1007\/978-3-319-72971-8_1"},{"key":"28_CR27","unstructured":"Vazhkudai, S.S., et al..: The design, deployment, and evaluation of the CORAL pre-exascale systems. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage, and Analysis, SC 2018, pp. 52:1\u201352:12. IEEE Press, Piscataway (2018). \nhttp:\/\/dl.acm.org\/citation.cfm?id=3291656.3291726"},{"issue":"10","key":"28_CR28","doi-asserted-by":"publisher","first-page":"2595","DOI":"10.1109\/TPDS.2013.222","volume":"25","author":"H Wang","year":"2014","unstructured":"Wang, H., Potluri, S., Bureddy, D., Rosales, C., Panda, D.K.: GPU-aware MPI on RDMA-enabled clusters: design, implementation and evaluation. IEEE Trans. Parallel Distrib. Syst. 25(10), 2595\u20132605 (2014). \nhttps:\/\/doi.org\/10.1109\/TPDS.2013.222","journal-title":"IEEE Trans. Parallel Distrib. Syst."}],"container-title":["Lecture Notes in Computer Science","High Performance Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-34356-9_28","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,5,13]],"date-time":"2020-05-13T16:29:22Z","timestamp":1589387362000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-34356-9_28"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019]]},"ISBN":["9783030343552","9783030343569"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-34356-9_28","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019]]},"assertion":[{"value":"3 December 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ISC High Performance","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on High Performance Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Frankfurt","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2019","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 June 2019","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 June 2019","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"34","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"supercomputing2019","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.isc-hpc.com\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Linklings","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"70","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"48","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"69% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4-5","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"n\/a","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}