{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,5,29]],"date-time":"2024-05-29T04:36:12Z","timestamp":1716957372836},"reference-count":32,"publisher":"Springer Science and Business Media LLC","issue":"16","license":[{"start":{"date-parts":[[2022,6,2]],"date-time":"2022-06-02T00:00:00Z","timestamp":1654128000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,6,2]],"date-time":"2022-06-02T00:00:00Z","timestamp":1654128000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"name":"Major Project on the Integration of Industry, Education and Research of Zhongshan","award":["210602103890051"],"award-info":[{"award-number":["210602103890051"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2022,11]]},"DOI":"10.1007\/s11227-022-04570-9","type":"journal-article","created":{"date-parts":[[2022,6,2]],"date-time":"2022-06-02T12:02:41Z","timestamp":1654171361000},"page":"18189-18208","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Memory-accelerated parallel method for multidimensional fast fourier implementation on GPU"],"prefix":"10.1007","volume":"78","author":[{"given":"Yichang","family":"Hu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lu","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cuixu","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,6,2]]},"reference":[{"key":"4570_CR1","doi-asserted-by":"publisher","unstructured":"Cheng S, Yu H-R, Inman D, Liao Q, Wu Q, Lin J (2020) Cube-towards an optimal scaling of cosmological n-body simulations. In: 2020 20th IEEE\/ACM International Symposium on Cluster, Cloud and Internet Computing (CCGRID), pp 685\u2013690. https:\/\/doi.org\/10.1109\/CCGrid49817.2020.00-22","DOI":"10.1109\/CCGrid49817.2020.00-22"},{"issue":"2","key":"4570_CR2","doi-asserted-by":"publisher","first-page":"215","DOI":"10.1016\/0043-1648(82)90178-8","volume":"83","author":"W Watson","year":"1982","unstructured":"Watson W, Spedding TA (1982) The time series modelling of non-gaussian engineering processes. Wear 83(2):215\u2013231. https:\/\/doi.org\/10.1016\/0043-1648(82)90178-8","journal-title":"Wear"},{"issue":"996","key":"4570_CR3","doi-asserted-by":"publisher","DOI":"10.1088\/1538-3873\/aaef0b","volume":"131","author":"CM Biwer","year":"2019","unstructured":"Biwer CM, Capano CD, De S, Cabero M, Brown DA, Nitz AH, Raymond V (2019) PyCBC inference: a python-based parameter estimation toolkit for compact binary coalescence signals. Science 131(996):024503. https:\/\/doi.org\/10.1088\/1538-3873\/aaef0b","journal-title":"Science"},{"issue":"1","key":"4570_CR4","doi-asserted-by":"publisher","first-page":"130","DOI":"10.1016\/S0010-4655(00)00049-7","volume":"130","author":"PD Haynes","year":"2000","unstructured":"Haynes PD, C\u00f4t\u00e9 M (2000) Parallel fast fourier transforms for electronic structure calculations. Comput Phys Commun 130(1):130\u2013136. https:\/\/doi.org\/10.1016\/S0010-4655(00)00049-7","journal-title":"Comput Phys Commun"},{"key":"4570_CR5","doi-asserted-by":"publisher","first-page":"76","DOI":"10.1016\/j.ejmp.2017.07.024","volume":"42","author":"P Despr\u00e9s","year":"2017","unstructured":"Despr\u00e9s P, Jia X (2017) A review of gpu-based medical image reconstruction. Physica Med 42:76\u201392. https:\/\/doi.org\/10.1016\/j.ejmp.2017.07.024","journal-title":"Physica Med"},{"issue":"4","key":"4570_CR6","first-page":"1","volume":"33","author":"BA Cipra","year":"2000","unstructured":"Cipra BA (2000) The best of the 20th century: editors name top 10 algorithms. SIAM News 33(4):1\u20132","journal-title":"SIAM News"},{"key":"4570_CR7","doi-asserted-by":"publisher","unstructured":"Frigo M, Johnson SG (1998) FFTW: an adaptive software architecture for the FFT. In: Proceedings of the 1998 IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP \u201998 (Cat. No.98CH36181), vol 3, pp 1381\u201313843. https:\/\/doi.org\/10.1109\/ICASSP.1998.681704","DOI":"10.1109\/ICASSP.1998.681704"},{"key":"4570_CR8","doi-asserted-by":"crossref","unstructured":"Frigo M, Johnson SG (1997) The fastest fourier transform in the west. mit-lcs-tr-728. In: The Proceedings of the 1998 International Conference on Acoustics, Speech, and Signal Processing, ICASSP \u201998","DOI":"10.21236\/ADA479065"},{"key":"4570_CR9","doi-asserted-by":"publisher","unstructured":"Nukada A, Sato K, Matsuoka S (2012) Scalable multi-gpu 3-d fft for tsubame 2.0 supercomputer. In: SC \u201912: Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis, pp 1\u201310. https:\/\/doi.org\/10.1109\/SC.2012.100","DOI":"10.1109\/SC.2012.100"},{"key":"4570_CR10","doi-asserted-by":"publisher","unstructured":"Gu L, Li X, Siegel J (2010) An empirically tuned 2D and 3D fft library on cuda gpu. In: Proceedings of the 24th ACM International Conference on Supercomputing. ICS \u201910, pp. 305\u2013314. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/1810085.1810127.","DOI":"10.1145\/1810085.1810127."},{"key":"4570_CR11","doi-asserted-by":"publisher","unstructured":"Schwaller B, Ramesh B, George AD (2017) Investigating ti keystone ii and quad-core arm cortex-a53 architectures for on-board space processing. In: 2017 IEEE High Performance Extreme Computing Conference (HPEC), pp 1\u20137. https:\/\/doi.org\/10.1109\/HPEC.2017.8091094","DOI":"10.1109\/HPEC.2017.8091094"},{"issue":"90","key":"4570_CR12","doi-asserted-by":"publisher","first-page":"297","DOI":"10.1090\/S0025-5718-1965-0178586-1","volume":"19","author":"JW Cooley","year":"1965","unstructured":"Cooley JW, Tukey JW (1965) An algorithm for the machine calculation of complex fourier series. Math Comput 19(90):297\u2013301","journal-title":"Math Comput"},{"key":"4570_CR13","doi-asserted-by":"publisher","unstructured":"Gentleman WM, Sande G (1966) Fast fourier transforms: For fun and profit. In: Proceedings of the November 7\u201310, 1966, Fall Joint Computer Conference. AFIPS \u201966 (Fall), pp 563\u2013578. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/1464291.1464352.","DOI":"10.1145\/1464291.1464352."},{"issue":"1","key":"4570_CR14","doi-asserted-by":"publisher","first-page":"45","DOI":"10.1016\/S0167-8191(84)90413-7","volume":"1","author":"PN Swarztrauber","year":"1984","unstructured":"Swarztrauber PN (1984) Fft algorithms for vector computers. Parallel Comput 1(1):45\u201363. https:\/\/doi.org\/10.1016\/S0167-8191(84)90413-7","journal-title":"Parallel Comput"},{"key":"4570_CR15","doi-asserted-by":"publisher","unstructured":"Luo Y, Li Y, Yang J, Ma L, Huang W, Xu B (2021) Optimization of the randomness extraction based on toeplitz matrix for high-speed qrng post-processing on gpu. In: 2021 13th International Conference on Communication Software and Networks (ICCSN), pp 261\u2013264. https:\/\/doi.org\/10.1109\/ICCSN52437.2021.9463613","DOI":"10.1109\/ICCSN52437.2021.9463613"},{"key":"4570_CR16","doi-asserted-by":"publisher","unstructured":"Zhao Z, Zhao Y (2018) The optimization of fft algorithm based with parallel computing on gpu. In: 2018 IEEE 3rd Advanced Information Technology, Electronic and Automation Control Conference (IAEAC), pp 2003\u20132007. https:\/\/doi.org\/10.1109\/IAEAC.2018.8577843","DOI":"10.1109\/IAEAC.2018.8577843"},{"issue":"1","key":"4570_CR17","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1002\/spe.2507","volume":"48","author":"P Nejedly","year":"2018","unstructured":"Nejedly P, Plesinger F, Halamek J, Jurak P (2018) Cudafilters: a signalplant library for gpu-accelerated fft and fir filtering. Softw Pract Exp 48(1):3\u20139. https:\/\/doi.org\/10.1002\/spe.2507","journal-title":"Softw Pract Exp"},{"key":"4570_CR18","doi-asserted-by":"publisher","unstructured":"Ogata Y, Endo T, Maruyama N, Matsuoka S (2008) An efficient, model-based cpu-gpu heterogeneous fft library. In: 2008 IEEE International Symposium on Parallel and Distributed Processing, pp 1\u201310. https:\/\/doi.org\/10.1109\/IPDPS.2008.4536163","DOI":"10.1109\/IPDPS.2008.4536163"},{"key":"4570_CR19","doi-asserted-by":"publisher","unstructured":"C\u0131lasun H, Resch S, Chowdhury ZI, Olson E, Zabihi M, Zhao Z, Peterson T, Wang J-P, Sapatnekar SS, Karpuzcu U (2020) CRAFFT: high resolution FFT accelerator in spintronic computational RAM. In: 2020 57th ACM\/IEEE Design Automation Conference (DAC), pp 1\u20136. https:\/\/doi.org\/10.1109\/DAC18072.2020.9218673","DOI":"10.1109\/DAC18072.2020.9218673"},{"issue":"10","key":"4570_CR20","doi-asserted-by":"publisher","first-page":"1953","DOI":"10.1109\/TVLSI.2018.2846688","volume":"26","author":"X Chen","year":"2018","unstructured":"Chen X, Lei Y, Lu Z, Chen S (2018) A variable-size fft hardware accelerator based on matrix transposition. IEEE Trans Very Large Scale Integr Syst 26(10):1953\u20131966. https:\/\/doi.org\/10.1109\/TVLSI.2018.2846688","journal-title":"IEEE Trans Very Large Scale Integr Syst"},{"key":"4570_CR21","doi-asserted-by":"publisher","unstructured":"Li Z, Jia H, Zhang Y, Chen T, Yuan L, Cao L, Wang X (2019) AutoFFT: a template-based FFT codes auto-generation framework for ARM and X86 CPUs. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis. SC \u201919. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3295500.3356138.","DOI":"10.1145\/3295500.3356138."},{"key":"4570_CR22","doi-asserted-by":"publisher","unstructured":"Ayala A, Tomov S, Luo X, Shaeik H, Haidar A, Bosilca G, Dongarra J (2019) Impacts of multi-gpu mpi collective communications on large fft computation. In: 2019 IEEE\/ACM Workshop on Exascale MPI (ExaMPI), pp 12\u201318. https:\/\/doi.org\/10.1109\/ExaMPI49596.2019.00007","DOI":"10.1109\/ExaMPI49596.2019.00007"},{"key":"4570_CR23","doi-asserted-by":"publisher","unstructured":"Chen S, Li X (2013) A hybrid gpu\/cpu fft library for large fft problems. In: 2013 IEEE 32nd International Performance Computing and Communications Conference (IPCCC), pp 1\u201310. https:\/\/doi.org\/10.1109\/PCCC.2013.6742796","DOI":"10.1109\/PCCC.2013.6742796"},{"key":"4570_CR24","unstructured":"Gholami A, Hill J, Malhotra D, Biros G (2015) AccFFT: a library for distributed-memory FFT on CPU and GPU architectures. arXiv preprint arXiv:1506.07933"},{"key":"4570_CR25","doi-asserted-by":"publisher","unstructured":"Cecka C (2017) Low communication fmm-accelerated fft on gpus. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis. SC \u201917. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3126908.3126919.","DOI":"10.1145\/3126908.3126919."},{"key":"4570_CR26","doi-asserted-by":"publisher","unstructured":"Markidis S, Chien SWD, Laure E, Peng IB, Vetter JS (2018) Nvidia tensor core programmability, performance amp; precision. In: 2018 IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW), pp 522\u2013531. https:\/\/doi.org\/10.1109\/IPDPSW.2018.00091","DOI":"10.1109\/IPDPSW.2018.00091"},{"key":"4570_CR27","doi-asserted-by":"publisher","unstructured":"Sorna A, Cheng X, D\u2019Azevedo E, Won K, Tomov S (2018) Optimizing the fast fourier transform using mixed precision on tensor core hardware. In: 2018 IEEE 25th International Conference on High Performance Computing Workshops (HiPCW), pp 3\u20137. https:\/\/doi.org\/10.1109\/HiPCW.2018.8634417","DOI":"10.1109\/HiPCW.2018.8634417"},{"key":"4570_CR28","unstructured":"Cheng X, Sorna A, D\u2019Azevedo E, Wong K, Tomov S (2018) Accelerating 2d fft: exploit gpu tensor cores through mixed-precision. In: The International Conference for High Performance Computing, Networking, Storage, and Analysis (SC\u201918), ACM Student Research Poster, Dallas, TX"},{"key":"4570_CR29","doi-asserted-by":"publisher","unstructured":"Durrani S, Chughtai MS, Dakkak A, Hwu W-m, Rauchwerger L (2021) FFT Blitz: The Tensor Cores Strike Back. In: Proceedings of the 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming. PPoPP \u201921, pp 488\u2013489. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3437801.3441623.","DOI":"10.1145\/3437801.3441623."},{"issue":"9","key":"4570_CR30","doi-asserted-by":"publisher","first-page":"1737","DOI":"10.1109\/TVLSI.2018.2825145","volume":"26","author":"T Abtahi","year":"2018","unstructured":"Abtahi T, Shea C, Kulkarni A, Mohsenin T (2018) Accelerating convolutional neural network with fft on embedded hardware. IEEE Trans Very Large Scale Integr Syst 26(9):1737\u20131749. https:\/\/doi.org\/10.1109\/TVLSI.2018.2825145","journal-title":"IEEE Trans Very Large Scale Integr Syst"},{"issue":"12","key":"4570_CR31","doi-asserted-by":"publisher","first-page":"19094","DOI":"10.1364\/OE.422266","volume":"29","author":"J Lee","year":"2021","unstructured":"Lee J, Kang H, Yeom H-J, Cheon S, Park J, Kim D (2021) Out-of-core gpu 2D-shift-fft algorithm for ultra-high-resolution hologram generation. Opt Express 29(12):19094\u201319112","journal-title":"Opt Express"},{"key":"4570_CR32","doi-asserted-by":"publisher","first-page":"120261","DOI":"10.1109\/ACCESS.2021.3108404","volume":"9","author":"H Kang","year":"2021","unstructured":"Kang H, Lee J, Kim D (2021) Hi-fft: Heterogeneous parallel in-place algorithm for large-scale 2D-fft. IEEE Access 9:120261\u2013120273. https:\/\/doi.org\/10.1109\/ACCESS.2021.3108404","journal-title":"IEEE Access"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-022-04570-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-022-04570-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-022-04570-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,17]],"date-time":"2022-11-17T15:19:28Z","timestamp":1668698368000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-022-04570-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,6,2]]},"references-count":32,"journal-issue":{"issue":"16","published-print":{"date-parts":[[2022,11]]}},"alternative-id":["4570"],"URL":"https:\/\/doi.org\/10.1007\/s11227-022-04570-9","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,6,2]]},"assertion":[{"value":"27 April 2022","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 June 2022","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}