{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T23:09:09Z","timestamp":1774998549122,"version":"3.50.1"},"reference-count":22,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018,11]]},"DOI":"10.1109\/pmbs.2018.8641632","type":"proceedings-article","created":{"date-parts":[[2019,2,14]],"date-time":"2019-02-14T23:44:06Z","timestamp":1550187846000},"page":"1-11","source":"Crossref","is-referenced-by-count":9,"title":["Improving MPI Reduction Performance for Manycore Architectures with OpenMP and Data Compression"],"prefix":"10.1109","author":[{"given":"Hongzhang","family":"Shan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Samuel","family":"Williams","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Calvin W.","family":"Johnson","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1177\/1094342014552086"},{"key":"ref11","year":"0","journal-title":"MPICH is a high performance and widely portable implementation of the message passing interface (mpi) standard"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/s10586-007-0012-0"},{"key":"ref13","first-page":"77","article-title":"Automatic MPI counter profiling of all users: First results on a CRAY T3E 900&#x2013;512","author":"rabenseifner","year":"1999","journal-title":"Proceedings of the Message Passing Interface Developer's Conference 1999 (MPIDC'99)"},{"key":"ref14","doi-asserted-by":"crossref","DOI":"10.1007\/978-3-540-30218-6_13","article-title":"More efficient reduction algorithms for non-powerof-two number of processors in message-passing parallel systems","author":"rabenseifner","year":"2004","journal-title":"Lecture Notes in Computer Science Proceedings of EuroPVM-MPI"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2017.15"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/2807591.2807618"},{"key":"ref17","article-title":"Optimization of MPI collectives on clusters of large-scale SMPs","author":"sistare","year":"1999","journal-title":"Proceedings of SC99 High Performance Networking and Computing"},{"key":"ref18","year":"0","journal-title":"High Performance SParse Communication for Machine Learning"},{"key":"ref19","article-title":"Improving the Performance of Collective Operations in MPICH","author":"thakur","year":"2003","journal-title":"Euro PVM\/MPI User's Group Meeting"},{"key":"ref4","doi-asserted-by":"crossref","first-page":"94","DOI":"10.1007\/978-3-540-87475-1_17","author":"hofmann","year":"2008","journal-title":"MPI Reduction Operations for Sparse Floating-point Data Recent Advances in Parallel Virtual Machine and Message Passing Interface"},{"key":"ref3","author":"faraj","year":"2005","journal-title":"Automatioc generation and tuning of MPI collective communication routines ICS"},{"key":"ref6","author":"johnson","year":"0","journal-title":"Bigstick A flexible configuration-interaction shell-model code"},{"key":"ref5","doi-asserted-by":"crossref","DOI":"10.1109\/SC.2014.33","article-title":"Maximizing Throughput on a Dragonfly Network","author":"jain","year":"2014","journal-title":"The International Conference for High Performance Computing Networking Storage and Analysis"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2010.5470853"},{"key":"ref7","first-page":"2761","volume":"184","author":"johnson","year":"2013","journal-title":"Factorization in large-scale many-body calculations Computer Physics Communications"},{"key":"ref2","year":"0","journal-title":"BIGSTICK"},{"key":"ref1","first-page":"685","article-title":"HPCToolkit: Tools for performance analysis of optimized parallel programs. Concurrency and Computation","volume":"22","author":"adhianto","year":"2010","journal-title":"Practice and Experience"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/HOTI.2013.26"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1177\/1094342005051521"},{"key":"ref22","doi-asserted-by":"crossref","first-page":"275","DOI":"10.1007\/978-3-642-15646-5_29","article-title":"Transparent neutral element elimination in mpi reduction operations","author":"tr?ff","year":"2010","journal-title":"Recent Advances in the Message Passing Interface"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2003.1213188"}],"event":{"name":"2018 IEEE\/ACM Performance Modeling, Benchmarking and Simulation of High Performance Computer Systems (PMBS)","location":"Dallas, TX, USA","start":{"date-parts":[[2018,11,12]]},"end":{"date-parts":[[2018,11,12]]}},"container-title":["2018 IEEE\/ACM Performance Modeling, Benchmarking and Simulation of High Performance Computer Systems (PMBS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8630816\/8641548\/08641632.pdf?arnumber=8641632","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,27]],"date-time":"2022-01-27T05:51:25Z","timestamp":1643262685000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8641632\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,11]]},"references-count":22,"URL":"https:\/\/doi.org\/10.1109\/pmbs.2018.8641632","relation":{},"subject":[],"published":{"date-parts":[[2018,11]]}}}