{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T23:45:58Z","timestamp":1783035958998,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":37,"publisher":"ACM","funder":[{"name":"NSF","award":["2316157"],"award-info":[{"award-number":["2316157"]}]},{"name":"NSF","award":["2221811"],"award-info":[{"award-number":["2221811"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,7,20]]},"DOI":"10.1145\/3731545.3731590","type":"proceedings-article","created":{"date-parts":[[2025,9,9]],"date-time":"2025-09-09T12:46:16Z","timestamp":1757421976000},"page":"1-13","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Parameterized Algorithms for Non-uniform All-to-all"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9563-6129","authenticated-orcid":false,"given":"Ke","family":"Fan","sequence":"first","affiliation":[{"name":"University of Illinois, Chicago, Chicago, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5343-414X","authenticated-orcid":false,"given":"Jens","family":"Domke","sequence":"additional","affiliation":[{"name":"RIKEN Center for Computational Science, Kobe, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8837-3116","authenticated-orcid":false,"given":"Seydou","family":"Ba","sequence":"additional","affiliation":[{"name":"RIKEN Center for Computational Science (R-CCS), Kobe, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-0418-9962","authenticated-orcid":false,"given":"Sidharth","family":"Kumar","sequence":"additional","affiliation":[{"name":"University of Illinois, Chicago, Chicago, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,9,9]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"MPICH Home Page. https:\/\/www.mpich.org."},{"key":"e_1_3_2_1_2_1","unstructured":"Argonne National Laboratory. 2024. Polaris | Argonne Leadership Computing Facility. https:\/\/www.alcf.anl.gov\/polaris."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.4851"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3078597.3078616"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3555819.3555825"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/71.642949"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW55747.2022.00014"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3624062.3624111"},{"key":"e_1_3_2_1_9_1","volume-title":"An algorithm for the machine computation of the complex fourier series, in mathematics of computation. April","author":"Cooley JW","year":"1965","unstructured":"JW Cooley and JW Tukey. 1965. An algorithm for the machine computation of the complex fourier series, in mathematics of computation. April (1965)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/2049662.2049663"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3502181.3531468"},{"key":"e_1_3_2_1_12_1","volume-title":"2021 IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW). IEEE, 965\u2013972","author":"Fan Ke","year":"2021","unstructured":"Ke Fan, Kristopher Micinski, Thomas Gilray, and Sidharth Kumar. 2021. Exploring MPI Collective I\/O and File-per-process I\/O for Checkpointing a Logical Inference Task. In 2021 IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW). IEEE, 965\u2013972."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.23919\/ISC.2024.10528936"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/1088149.1088202"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-30218-6_19"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/2966884.2966918"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3446804.3446855"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"William Gropp and Ewing Lusk. 1996. User's Guide for mpich a Portable Implementation of MPI.","DOI":"10.2172\/378911"},{"key":"e_1_3_2_1_19_1","unstructured":"Adrian Jackson and Stephen Booth. 2004. Planned AlltoAllv a Cluster Approach. (2004)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.4964"},{"key":"e_1_3_2_1_21_1","volume-title":"Optimised allgatherv, reduce_scatter and allreduce communication in message-passing systems. arXiv preprint arXiv:2006.13112","author":"Jocksch Andreas","year":"2020","unstructured":"Andreas Jocksch, Noe Ohana, Emmanuel Lanti, Vasileios Karakasis, and Laurent Villard. 2020. Optimised allgatherv, reduce_scatter and allreduce communication in message-passing systems. arXiv preprint arXiv:2006.13112 (2020)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.5555\/3433701.3433714"},{"key":"e_1_3_2_1_23_1","volume-title":"International Conference on High Performance Computing, Data, and Analytics (HiPC). IEEE.","author":"Kumar Sidharth","year":"2019","unstructured":"Sidharth Kumar and Thomas Gilray. 2019. Distributed Relational Algebra at Scale. In International Conference on High Performance Computing, Data, and Analytics (HiPC). IEEE."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-50743-5_15"},{"key":"e_1_3_2_1_25_1","volume-title":"2016 IEEE International Conference on Big Data (Big Data). IEEE, 56\u201365","author":"Moustafa Walaa Eldin","year":"2016","unstructured":"Walaa Eldin Moustafa, Vicky Papavasileiou, Ken Yocum, and Alin Deutsch. 2016. Datalography: Scaling datalog graph analytics on graph processing systems. In 2016 IEEE International Conference on Big Data (Big Data). IEEE, 56\u201365."},{"key":"e_1_3_2_1_26_1","volume-title":"2022 IEEE 29th International Conference on High Performance Computing, Data and Analytics Workshop (HiPCW). IEEE, 20\u201327","author":"Netterville Naeris","year":"2022","unstructured":"Naeris Netterville, Ke Fan, Sidharth Kumar, and Thomas Gilray. 2022. A Visual Guide to MPI All-to-all. In 2022 IEEE 29th International Conference on High Performance Computing, Data and Analytics Workshop (HiPCW). IEEE, 20\u201327."},{"key":"e_1_3_2_1_27_1","unstructured":"NVIDIA Corporation. 2022. Multinode Multi-GPU: Using NVIDIA cuFFTMp FFTs at Scale. https:\/\/developer.nvidia.com\/blog\/multinode-multi-gpu-using-nvidia-cufftmp-ffts-at-scale\/."},{"key":"e_1_3_2_1_28_1","volume-title":"2021 IEEE\/ACM 6th International Workshop on Extreme Scale Programming Models and Middleware (ESPM2). IEEE, 1\u20139.","author":"Patel Sarthak","year":"2021","unstructured":"Sarthak Patel, Bhrugu Dave, Smit Kumbhani, Mihir Desai, Sidharth Kumar, and Bhaskar Chaudhury. 2021. Scalable parallel algorithm for fast computation of Transitive Closure of Graphs on Shared Memory Architectures. In 2021 IEEE\/ACM 6th International Workshop on Extreme Scale Programming Models and Middleware (ESPM2). IEEE, 1\u20139."},{"key":"e_1_3_2_1_29_1","unstructured":"Martin Plummer and Keith Refson. 2004. An lpar-customized mpi alltoallv for the materials science code castep. Technical Report EPCC (Edinburgh Parallel Computing Centre) (2004)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.5555\/3433701.3433763"},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of EuroMPI 2021","author":"Taru Doodi","year":"2021","unstructured":"Doodi Taru, Nusrat Islam, Gengbin Zheng, Rubasri Kalidas, Akhil Langer, and Maria Garzaran. 2021. High Radix Collective Algorithms. Proceedings of EuroMPI 2021 (2021)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342005051521"},{"key":"e_1_3_2_1_33_1","volume-title":"Proceedings of the 28th ACM international conference on Supercomputing.","author":"Tr\u00e4ff Jesper Larsson","year":"2014","unstructured":"Jesper Larsson Tr\u00e4ff, Antoine Rougier, and Sascha Hunold. 2014. Implementing a classic: Zero-copy all-to-all communication with MPI datatypes. In Proceedings of the 28th ACM international conference on Supercomputing."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2012.28"},{"key":"e_1_3_2_1_35_1","volume-title":"Hans De Raedt, and Kristel Michielsen","author":"Willsch Dennis","year":"2023","unstructured":"Dennis Willsch, Madita Willsch, Fengping Jin, Hans De Raedt, and Kristel Michielsen. 2023. Large-Scale Simulation of Shor's Quantum Factoring Algorithm. Mathematics 11, 19 (2023)."},{"key":"e_1_3_2_1_36_1","volume-title":"2013 13th IEEE\/ACM International Symposium on Cluster, Cloud, and Grid Computing. IEEE, 369\u2013376","author":"Xu Cong","year":"2013","unstructured":"Cong Xu, Manjunath Gorentla Venkata, Richard L Graham, Yandong Wang, Zhuo Liu, and Weikuan Yu. 2013. Sloavx: Scalable logarithmic alltoallv algorithm for hierarchical multicore systems. In 2013 13th IEEE\/ACM International Symposium on Cluster, Cloud, and Grid Computing. IEEE, 369\u2013376."},{"key":"e_1_3_2_1_37_1","volume-title":"Accelerating MPI AllReduce Communication with Efficient GPU-Based Compression Schemes on Modern GPU Clusters. In ISC HIGH PERFORMANCE","author":"Zhou Q.","year":"2024","unstructured":"Q. Zhou, B. Ramesh, A. Shafi, M. Abduljabbar, H. Subramoni, and D. Panda. 2024. Accelerating MPI AllReduce Communication with Efficient GPU-Based Compression Schemes on Modern GPU Clusters. In ISC HIGH PERFORMANCE 2024."}],"event":{"name":"HPDC '25: 34th International Symposium on High-Performance Parallel and Distributed Computing","location":"University of Notre Dame Conference Facilities Notre Dame IN USA","acronym":"HPDC '25","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 34th International Symposium on High-Performance Parallel and Distributed Computing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3731545.3731590","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,9]],"date-time":"2025-09-09T12:46:51Z","timestamp":1757422011000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731545.3731590"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,20]]},"references-count":37,"alternative-id":["10.1145\/3731545.3731590","10.1145\/3731545"],"URL":"https:\/\/doi.org\/10.1145\/3731545.3731590","relation":{},"subject":[],"published":{"date-parts":[[2025,7,20]]},"assertion":[{"value":"2025-09-09","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}