{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,30]],"date-time":"2025-12-30T15:33:39Z","timestamp":1767108819133},"reference-count":27,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2012,10,25]],"date-time":"2012-10-25T00:00:00Z","timestamp":1351123200000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/2.0"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2013,3]]},"DOI":"10.1007\/s11227-012-0836-0","type":"journal-article","created":{"date-parts":[[2012,10,24]],"date-time":"2012-10-24T05:18:09Z","timestamp":1351055889000},"page":"897-918","source":"Crossref","is-referenced-by-count":17,"title":["Adaptive fast multipole methods on the GPU"],"prefix":"10.1007","volume":"63","author":[{"given":"Anders","family":"Goude","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stefan","family":"Engblom","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2012,10,25]]},"reference":[{"issue":"3","key":"836_CR1","doi-asserted-by":"crossref","first-page":"773","DOI":"10.1137\/S1064827593272031","volume":"17","author":"S Aluru","year":"1996","unstructured":"Aluru S (1996) Greengard\u2019s N-body algorithm is not order N. SIAM J Sci Comput 17(3):773\u2013776. doi: 10.1137\/S1064827593272031","journal-title":"SIAM J Sci Comput"},{"key":"836_CR2","series-title":"Series in discrete mathematics and theoretical computer science","volume-title":"Parallel algorithms","author":"G Blelloch","year":"1997","unstructured":"Blelloch G, Narlikar G (1997) A practical comparison of N-body algorithms. In: Parallel algorithms. Series in discrete mathematics and theoretical computer science, vol 30"},{"key":"836_CR3","doi-asserted-by":"crossref","first-page":"75","DOI":"10.1016\/B978-0-12-384988-5.00006-1","volume-title":"GPU computing Gems Emerald edition","author":"M Burtscher","year":"2011","unstructured":"Burtscher M, Pingali K (2011) An efficient CUDA implementation of the tree-based Barnes Hut n-body algorithm. In: Hwu W-M (ed) GPU computing Gems Emerald edition, vol 6. Morgan Kaufmann\/Elsevier, San Mateo, pp 75\u201392"},{"issue":"4","key":"836_CR4","doi-asserted-by":"crossref","first-page":"669","DOI":"10.1137\/0909044","volume":"9","author":"J Carrier","year":"1988","unstructured":"Carrier J, Greengard L, Rokhlin V (1988) A fast adaptive multipole algorithm for particle simulations. SIAM J Sci Stat Comput 9(4):669\u2013686. doi: 10.1137\/0909044","journal-title":"SIAM J Sci Stat Comput"},{"key":"836_CR5","unstructured":"Cipra BA (2000) The best of the 20th century: editors name top 10 algorithms. SIAM News 33(4)"},{"issue":"4","key":"836_CR6","doi-asserted-by":"crossref","first-page":"403","DOI":"10.1002\/nme.2972","volume":"85","author":"FA Cruz","year":"2011","unstructured":"Cruz FA, Knepley MG, Barba LA (2011) Petfmm\u2013a dynamically load-balancing parallel fast multipole library. Int J Numer Methods Eng 85(4):403\u2013428. doi: 10.1002\/nme.2972","journal-title":"Int J Numer Methods Eng"},{"issue":"10","key":"836_CR7","doi-asserted-by":"crossref","first-page":"1096","DOI":"10.1016\/j.apnum.2011.06.011","volume":"61","author":"S Engblom","year":"2011","unstructured":"Engblom S (2011) On well-separated sets and fast multipole methods. Appl Numer Math 61(10):1096\u20131102. doi: 10.1016\/j.apnum.2011.06.011","journal-title":"Appl Numer Math"},{"key":"836_CR8","volume-title":"CUDA application design and development","author":"R Farber","year":"2011","unstructured":"Farber R (2011) CUDA application design and development, 1st edn. Morgan Kaufmann, San Francisco","edition":"1"},{"issue":"2","key":"836_CR9","doi-asserted-by":"crossref","first-page":"325","DOI":"10.1016\/0021-9991(87)90140-9","volume":"73","author":"L Greengard","year":"1987","unstructured":"Greengard L, Rokhlin V (1987) A fast algorithm for particle simulations. J Comput Phys 73(2):325\u2013348. doi: 10.1016\/0021-9991(87)90140-9","journal-title":"J Comput Phys"},{"key":"836_CR10","series-title":"Texts in computational science and engineering","volume-title":"Numerical simulation in molecular dynamics","author":"M Griebel","year":"2007","unstructured":"Griebel M, Knapek S, Zumbusch G (2007) Numerical simulation in molecular dynamics. Texts in computational science and engineering, vol 5. Springer, Berlin"},{"key":"836_CR11","series-title":"Elsevier series in electromagnetism","volume-title":"Fast multipole methods for the Helmholtz equation in three dimensions","author":"NA Gumerov","year":"2004","unstructured":"Gumerov NA, Duraiswami R (2004) Fast multipole methods for the Helmholtz equation in three dimensions. Elsevier series in electromagnetism. Elsevier, Oxford"},{"issue":"18","key":"836_CR12","doi-asserted-by":"crossref","first-page":"8290","DOI":"10.1016\/j.jcp.2008.05.023","volume":"227","author":"NA Gumerov","year":"2008","unstructured":"Gumerov NA, Duraiswami R (2008) Fast multipole methods on graphics processors. J Comput Phys 227(18):8290\u20138313. doi: 10.1016\/j.jcp.2008.05.023","journal-title":"J Comput Phys"},{"key":"836_CR13","first-page":"851","volume-title":"GPU gems 3","author":"M Harris","year":"2007","unstructured":"Harris M, Sengupta S, Owens JD (2007) Parallel prefix sum (scan) with CUDA. In: Nguyen H (ed) GPU gems 3. Addison Wesley, Reading, pp 851\u2013876. chapter 39"},{"issue":"6","key":"836_CR14","doi-asserted-by":"crossref","first-page":"1804","DOI":"10.1137\/S106482759630989X","volume":"19","author":"T Hrycak","year":"1998","unstructured":"Hrycak T, Rokhlin V (1998) An improved fast multipole algorithm for potential fields. SIAM J Sci Comput 19(6):1804\u20131826. doi: 10.1137\/S106482759630989X","journal-title":"SIAM J Sci Comput"},{"key":"836_CR15","doi-asserted-by":"crossref","DOI":"10.1017\/CBO9780511605345","volume-title":"Fast multipole boundary element method: theory and applications in engineering","author":"Y Liu","year":"2009","unstructured":"Liu Y (2009) Fast multipole boundary element method: theory and applications in engineering. Cambridge University Press, Cambridge"},{"key":"836_CR16","unstructured":"NVIDIA\u2019s next generation CUDA compute architecture: Fermi. NVIDIA (2009) Available at http:\/\/www.nvidia.com\/content\/PDF\/fermi_white_papers\/NVIDIA_Fermi_Compute_Architecture_Whitepaper.pdf"},{"key":"836_CR17","unstructured":"CUDA C programming guide. NVIDIA (2012). Version 4.2. Available at http:\/\/developer.download.nvidia.com\/compute\/DevZone\/docs\/html\/C\/doc\/CUDACProgrammingGuide.pdf"},{"issue":"4","key":"836_CR18","doi-asserted-by":"crossref","first-page":"926","DOI":"10.1016\/j.cpc.2010.12.029","volume":"182","author":"DC Rapaport","year":"2011","unstructured":"Rapaport DC (2011) Enhanced molecular dynamics performance with a programmable graphics processor. Comput Phys Commun 182(4):926\u2013934. doi: 10.1016\/j.cpc.2010.12.029","journal-title":"Comput Phys Commun"},{"key":"836_CR19","series-title":"Addison-Wesley series in computer science.","volume-title":"Algorithms in C","author":"R Sedgewick","year":"1990","unstructured":"Sedgewick R (1990) Algorithms in C. Addison-Wesley series in computer science. Addison-Wesley, Reading"},{"issue":"1","key":"836_CR20","doi-asserted-by":"crossref","first-page":"732","DOI":"10.1016\/j.jcp.2007.04.033","volume":"226","author":"B Shanker","year":"2007","unstructured":"Shanker B, Huang H (2007) Accelerated Cartesian expansions\u2014a fast method for computing of potentials of the form R \u2212\u03bd for all real \u03bd. J Comput Phys 226(1):732\u2013753. doi: 10.1016\/j.jcp.2007.04.033","journal-title":"J Comput Phys"},{"issue":"5","key":"836_CR21","doi-asserted-by":"crossref","first-page":"2675","DOI":"10.1137\/070681727","volume":"30","author":"H Sundar","year":"2008","unstructured":"Sundar H, Sampath RS, Biros G (2008) Bottom-up construction and 2:1 balance refinement of linear octrees in parallel. SIAM J Sci Comput 30(5):2675\u20132708. doi: 10.1137\/070681727","journal-title":"SIAM J Sci Comput"},{"issue":"2","key":"836_CR22","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1016\/j.newast.2011.07.001","volume":"17","author":"A Tanikawa","year":"2012","unstructured":"Tanikawa A, Yoshikawa K, Okamoto T, Nitadori K (2012) N-body simulation for self-gravitating collisional systems with a new SIMD instruction set extension to the x86 architecture, advanced vector extensions. New Astron 17(2):82\u201392. doi: 10.1016\/j.newast.2011.07.001","journal-title":"New Astron"},{"key":"836_CR23","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/IPDPS.2009.5161038","volume-title":"Proceedings of the 2009 IEEE international parallel and distributed processing symposium","author":"M Vikram","year":"2009","unstructured":"Vikram M, Baczewzki A, Shanker B, Aluru S (2009) Parallel accelerated Cartesian expansions for particle dynamics simulations. In: Proceedings of the 2009 IEEE international parallel and distributed processing symposium, pp 1\u201311. doi: 10.1109\/IPDPS.2009.5161038"},{"key":"836_CR24","unstructured":"Vuduc R, Chandramowlishwaran A, Choi J, Guney M, Shringarpure A (2010) On the limits of GPU acceleration. In: Proceedings of the 2nd USENIX conference on hot topics in parallelism, HotPar\u201910. Berkeley, CA, USA, USENIX Association"},{"key":"836_CR25","doi-asserted-by":"crossref","first-page":"235","DOI":"10.1109\/ISPASS.2010.5452013","volume-title":"2010 IEEE international symposium on performance analysis of systems software (ISPASS)","author":"H Wong","year":"2010","unstructured":"Wong H, Papadopoulou M-M, Sadooghi-Alvandi M, Moshovos A (2010) Demystifying GPU microarchitecture through microbenchmarking. In: 2010 IEEE international symposium on performance analysis of systems software (ISPASS), pp 235\u2013246. doi: 10.1109\/ISPASS.2010.5452013"},{"key":"836_CR26","doi-asserted-by":"crossref","first-page":"113","DOI":"10.1016\/B978-0-12-384988-5.00009-7","volume-title":"GPU computing Gems Emerald edition","author":"R Yokota","year":"2011","unstructured":"Yokota R, Barba L (2011) Treecode and fast multipole method for N-body simulation with CUDA. In: Hwu W-M (ed) GPU computing Gems Emerald edition Morgan Kaufmann\/Elsevier, San Mateo, pp 113\u2013132"},{"key":"836_CR27","author":"R Yokota","year":"2012","unstructured":"Yokota R, Barba L (2012) A tuned and scalable Fast Multipole Method as a preeminent algorithm for exascale systems. Int J High Perform Comput Appl. doi: 10.1177\/1094342011429952","journal-title":"Int J High Perform Comput Appl"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-012-0836-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11227-012-0836-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-012-0836-0","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,3,26]],"date-time":"2019-03-26T19:15:05Z","timestamp":1553627705000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11227-012-0836-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2012,10,25]]},"references-count":27,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2013,3]]}},"alternative-id":["836"],"URL":"https:\/\/doi.org\/10.1007\/s11227-012-0836-0","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2012,10,25]]}}}