{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,14]],"date-time":"2025-05-14T04:15:52Z","timestamp":1747196152109,"version":"3.40.5"},"publisher-location":"Cham","reference-count":19,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031733697"},{"type":"electronic","value":"9783031733703"}],"license":[{"start":{"date-parts":[[2024,9,25]],"date-time":"2024-09-25T00:00:00Z","timestamp":1727222400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,9,25]],"date-time":"2024-09-25T00:00:00Z","timestamp":1727222400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73370-3_5","type":"book-chapter","created":{"date-parts":[[2024,9,24]],"date-time":"2024-09-24T21:02:58Z","timestamp":1727211778000},"page":"75-88","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Improved MPI Collectives for\u00a03D-FFT"],"prefix":"10.1007","author":[{"given":"Yuang","family":"Yan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Natasha","family":"Kuk","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ryan E.","family":"Grant","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,9,25]]},"reference":[{"key":"5_CR1","doi-asserted-by":"publisher","unstructured":"Aseeri, S.A., Gopal\u00a0Chatterjee, A., Verma, M.K., Keyes, D.E.: A scheduling policy to save 10% of communication time in parallel fast Fourier transform. Concurrency Comput. Pract. Experience 35(15), e6508 (2023). https:\/\/doi.org\/10.1002\/cpe.6508. https:\/\/onlinelibrary.wiley.com\/doi\/abs\/10.1002\/cpe.6508","DOI":"10.1002\/cpe.6508"},{"key":"5_CR2","doi-asserted-by":"publisher","unstructured":"Ayala, A., Tomov, S., Stoyanov, M., Haidar, A., Dongarra, J.: Accelerating multi - process communication for parallel 3-D FFT. In: 2021 Workshop on Exascale MPI (ExaMPI), pp. 46\u201353 (2021). https:\/\/doi.org\/10.1109\/ExaMPI54564.2021.00011","DOI":"10.1109\/ExaMPI54564.2021.00011"},{"key":"5_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"262","DOI":"10.1007\/978-3-030-50371-0_19","volume-title":"Computational Science \u2013 ICCS 2020","author":"A Ayala","year":"2020","unstructured":"Ayala, A., Tomov, S., Haidar, A., Dongarra, J.: heFFTe: highly efficient FFT for exascale. In: Krzhizhanovskaya, V.V., et al. (eds.) ICCS 2020. LNCS, vol. 12137, pp. 262\u2013275. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-50371-0_19"},{"key":"5_CR4","doi-asserted-by":"publisher","unstructured":"Carpen-Amarie, A., Hunold, S., Tr\u00e4ff, J.L.: On the expected and observed communication performance with MPI derived datatypes. In: Proceedings of the 23rd European MPI Users\u2019 Group Meeting, EuroMPI 2016, pp. 108\u2013120. Association for Computing Machinery, New York, NY, USA (2016). https:\/\/doi.org\/10.1145\/2966884.2966905","DOI":"10.1145\/2966884.2966905"},{"key":"5_CR5","doi-asserted-by":"publisher","unstructured":"Chatterjee, A.G., Verma, M.K., Kumar, A., Samtaney, R., Hadri, B., Khurram, R.: Scaling of a fast Fourier transform and a pseudo-spectral fluid solver up to 196608 cores. J. Parallel Distrib. Comput. 113, 77\u201391 (2018). https:\/\/doi.org\/10.1016\/j.jpdc.2017.10.014. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0743731517302903","DOI":"10.1016\/j.jpdc.2017.10.014"},{"key":"5_CR6","doi-asserted-by":"crossref","unstructured":"Cooley, J.W., Tukey, J.W.: An algorithm for the machine calculation of complex Fourier series. Math. Comput. 19, 297\u2013301 (1965). http:\/\/cr.yp.to\/bib\/entries.html#1965\/cooley","DOI":"10.1090\/S0025-5718-1965-0178586-1"},{"key":"5_CR7","doi-asserted-by":"publisher","unstructured":"Czechowski, K., Battaglino, C., McClanahan, C., Iyer, K., Yeung, P.K., Vuduc, R.: On the communication complexity of 3D FFTs and its implications for exascale. In: Proceedings of the 26th ACM International Conference on Supercomputing, ICS 2012, pp. 205\u2013214. Association for Computing Machinery, New York, NY, USA (2012). https:\/\/doi.org\/10.1145\/2304576.2304604","DOI":"10.1145\/2304576.2304604"},{"key":"5_CR8","doi-asserted-by":"publisher","unstructured":"Dalcin, L., Mortensen, M., Keyes, D.E.: Fast parallel multidimensional FFT using advanced MPI. J. Parallel Distributed Comput. 128, 137\u2013150 (2019). https:\/\/doi.org\/10.1016\/j.jpdc.2019.02.006. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S074373151830306X","DOI":"10.1016\/j.jpdc.2019.02.006"},{"key":"5_CR9","doi-asserted-by":"crossref","unstructured":"Frigo, M., Johnson, S.G.: The design and implementation of FFTW3. Proc. IEEE 93(2), 216\u2013231 (2005). Special issue on \u201cProgram Generation, Optimization, and Platform Adaptation\u201d","DOI":"10.1109\/JPROC.2004.840301"},{"key":"5_CR10","unstructured":"Gholami, A., Hill, J., Malhotra, D., Biros, G.: AccFFT: a library for distributed-memory FFT on CPU and GPU architectures (2016)"},{"key":"5_CR11","unstructured":"Gropp, W., Lusk, E., Swider, D.: Improving the performance of MPI derived datatypes. In: Proceedings of the Third MPI Developer\u2019s and User\u2019s Conference, pp. 25\u201330. Citeseer (1999)"},{"key":"5_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"132","DOI":"10.1007\/978-3-642-15646-5_14","volume-title":"Recent Advances in the Message Passing Interface","author":"T Hoefler","year":"2010","unstructured":"Hoefler, T., Gottlieb, S.: Parallel zero-copy algorithms for fast Fourier transform and conjugate gradient using MPI datatypes. In: Keller, R., Gabriel, E., Resch, M., Dongarra, J. (eds.) EuroMPI 2010. LNCS, vol. 6305, pp. 132\u2013141. Springer, Heidelberg (2010). https:\/\/doi.org\/10.1007\/978-3-642-15646-5_14"},{"issue":"10","key":"5_CR13","doi-asserted-by":"publisher","first-page":"2627","DOI":"10.1109\/TPDS.2013.234","volume":"25","author":"J Jenkins","year":"2013","unstructured":"Jenkins, J., Dinan, J., Balaji, P., Peterka, T., Samatova, N.F., Thakur, R.: Processing MPI derived datatypes on noncontiguous GPU-resident data. IEEE Trans. Parallel Distrib. Syst. 25(10), 2627\u20132637 (2013). https:\/\/doi.org\/10.1109\/TPDS.2013.234","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"5_CR14","doi-asserted-by":"publisher","unstructured":"Jung, J., Kobayashi, C., Imamura, T., Sugita, Y.: Parallel implementation of 3D FFT with volumetric decomposition schemes for efficient molecular dynamics simulations. Comput. Phys. Commun. 200, 57\u201365 (2016). https:\/\/doi.org\/10.1016\/j.cpc.2015.10.024. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0010465515004063","DOI":"10.1016\/j.cpc.2015.10.024"},{"key":"5_CR15","doi-asserted-by":"publisher","unstructured":"Pekurovsky, D.: P3DFFT: a framework for parallel computations of Fourier transforms in three dimensions. SIAM J. Sci. Comput. 34(4), C192\u2013C209 (2012). https:\/\/doi.org\/10.1137\/11082748X","DOI":"10.1137\/11082748X"},{"key":"5_CR16","doi-asserted-by":"crossref","unstructured":"Sewell, A., Fan, K., Shovon, A.R., Dyken, L., Kumar, S., Petruzza, S.: Bruck algorithm performance analysis for multi-GPU all-to-all communication. In: Proceedings of the International Conference on High Performance Computing in Asia-Pacific Region, pp. 127\u2013133 (2024)","DOI":"10.1145\/3635035.3635047"},{"key":"5_CR17","doi-asserted-by":"publisher","unstructured":"Suresh, K.K., et al.: Network assisted non-contiguous transfers for GPU-aware MPI libraries. In: 2022 IEEE Symposium on High-Performance Interconnects (HOTI), pp. 13\u201320 (2022). https:\/\/doi.org\/10.1109\/HOTI55740.2022.00018","DOI":"10.1109\/HOTI55740.2022.00018"},{"key":"5_CR18","doi-asserted-by":"publisher","unstructured":"Turnes, C.K., Romberg, J.: Spiral FFT: an efficient method for 3-D FFTs on spiral MRI contours. In: 2010 IEEE International Conference on Image Processing, pp. 617\u2013620 (2010). https:\/\/doi.org\/10.1109\/ICIP.2010.5653254","DOI":"10.1109\/ICIP.2010.5653254"},{"key":"5_CR19","doi-asserted-by":"publisher","unstructured":"Xiong, Q., Bangalore, P.V., Skjellum, A., Herbordt, M.: MPI derived datatypes: performance and portability issues. In: Proceedings of the 25th European MPI Users\u2019 Group Meeting, EuroMPI 2018. Association for Computing Machinery, New York, NY, USA (2018). https:\/\/doi.org\/10.1145\/3236367.3236378","DOI":"10.1145\/3236367.3236378"}],"container-title":["Lecture Notes in Computer Science","Recent Advances in the Message Passing Interface"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73370-3_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,13]],"date-time":"2025-05-13T11:51:45Z","timestamp":1747137105000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73370-3_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9,25]]},"ISBN":["9783031733697","9783031733703"],"references-count":19,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73370-3_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,9,25]]},"assertion":[{"value":"25 September 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}