{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T22:17:50Z","timestamp":1740176270457,"version":"3.37.3"},"reference-count":40,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2018,7,1]],"date-time":"2018-07-01T00:00:00Z","timestamp":1530403200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2018,7,1]],"date-time":"2018-07-01T00:00:00Z","timestamp":1530403200000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2018,7,1]],"date-time":"2018-07-01T00:00:00Z","timestamp":1530403200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2018,7,1]],"date-time":"2018-07-01T00:00:00Z","timestamp":1530403200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"US National Science Foundation","award":["0905046"],"award-info":[{"award-number":["0905046"]}]},{"name":"NSF CCF","award":["1539567"],"award-info":[{"award-number":["1539567"]}]},{"name":"LEQSF","award":["2015-16-ENH-TR-10"],"award-info":[{"award-number":["2015-16-ENH-TR-10"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Multi-Scale Comp. Syst."],"published-print":{"date-parts":[[2018,7,1]]},"DOI":"10.1109\/tmscs.2017.2705139","type":"journal-article","created":{"date-parts":[[2017,5,17]],"date-time":"2017-05-17T18:29:12Z","timestamp":1495045752000},"page":"477-490","source":"Crossref","is-referenced-by-count":1,"title":["Thoroughly Exploring GPU Buffering Options for Stencil Code by Using an Efficiency Measure and a Performance Model"],"prefix":"10.1109","volume":"4","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8959-6995","authenticated-orcid":false,"given":"Yue","family":"Hu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"David M.","family":"Koppelman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Steven Robert","family":"Brandt","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","first-page":"1","article-title":"Highly optimized full GPU-acceleration of non-hydrostatic weather model SCALE-LES","author":"wahib","year":"2013","journal-title":"Proc IEEE Int Conf Cluster Comput"},{"doi-asserted-by":"publisher","key":"ref38","DOI":"10.1109\/SC.2014.21"},{"doi-asserted-by":"publisher","key":"ref33","DOI":"10.1145\/2400682.2400690"},{"doi-asserted-by":"publisher","key":"ref32","DOI":"10.1145\/2400682.2400718"},{"key":"ref31","article-title":"Stencil computation optimization and auto-tuning on state-of-the-art multicore architectures","author":"datta","year":"2008","journal-title":"Proc ACM\/IEEE Conf Supercomputing"},{"key":"ref30","first-page":"311","article-title":"High-performance code generation for stencil computations on GPU architectures","author":"holewinski","year":"2012","journal-title":"Proc 26th ACM Int Conf Supercomput"},{"year":"2014","author":"usabiaga","article-title":"Minimal models for finite particles in fluctuating hydrodynamics","key":"ref37"},{"year":"2013","article-title":"Whitepaper: NVIDIA&#x2019;s next generation CUDA compute architecture: Kepler GK110","key":"ref36"},{"key":"ref35","first-page":"266","article-title":"Autotuning stencil-based computations on GPUs","author":"mametjanov","year":"2012","journal-title":"Proc IEEE Int Conf Cluster Comput"},{"doi-asserted-by":"publisher","key":"ref34","DOI":"10.1109\/TPDS.2003.1233716"},{"doi-asserted-by":"publisher","key":"ref10","DOI":"10.1109\/SC.2014.70"},{"year":"2016","author":"hu","article-title":"A performance model and optimization strategies for automatic GPU code generation of PDE systems described by a domain-specific language","key":"ref40"},{"key":"ref11","first-page":"79","article-title":"3D finite difference computation on GPUs using CUDA","author":"micikevicius","year":"2009","journal-title":"Proc 2nd Workshop General Purpose Process Graph Process Units"},{"doi-asserted-by":"publisher","key":"ref12","DOI":"10.1002\/cpe.3351"},{"doi-asserted-by":"publisher","key":"ref13","DOI":"10.1111\/j.1365-246X.2010.04616.x"},{"key":"ref14","first-page":"1","article-title":"Parallel data-locality aware stencil computations on modern micro-architectures","author":"christen","year":"2009","journal-title":"Proc IEEE Int Symp Parallel Distrib Process"},{"key":"ref15","article-title":"Understanding stencil code performance on multicore architectures","author":"rahman","year":"2011","journal-title":"Proc 8th ACM Int Conf Comput Frontiers"},{"doi-asserted-by":"publisher","key":"ref16","DOI":"10.1109\/CGO.2015.7054184"},{"doi-asserted-by":"publisher","key":"ref17","DOI":"10.1145\/2370816.2370858"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.1145\/2597652.2597685"},{"doi-asserted-by":"publisher","key":"ref19","DOI":"10.1109\/MICRO.2012.18"},{"key":"ref28","first-page":"24","article-title":"A performance model for memory bandwidth constrained applications on graphics engines","author":"ma","year":"2012","journal-title":"Proc IEEE 23rd Int Conf Appl -Specific Syst Archit Processors"},{"key":"ref4","doi-asserted-by":"crossref","first-page":"214","DOI":"10.1145\/1995896.1995932","article-title":"Mint: Realizing CUDA performance in 3D stencil methods with annotated C","author":"unat","year":"2011","journal-title":"Proc Int Conf Supercomputing"},{"doi-asserted-by":"publisher","key":"ref27","DOI":"10.1145\/1837853.1693470"},{"key":"ref3","first-page":"1","article-title":"Physis: An implicitly parallel programming model for stencil computations on large-scale GPU-accelerated supercomputers","author":"maruyama","year":"2011","journal-title":"Proc Int Conf High Performance Comput Netw Storage Anal"},{"key":"ref6","first-page":"1","article-title":"From physics model to results: An optimizing framework for cross-architecture code generation","volume":"21","author":"blazewicz","year":"2013","journal-title":"Sci Program"},{"doi-asserted-by":"publisher","key":"ref29","DOI":"10.1145\/2751205.2751240"},{"key":"ref5","first-page":"1","article-title":"PATUS for convenient high-performance stencils: Evaluation in earthquake simulations","author":"christen","year":"2012","journal-title":"Proc Int Conf High Performance Comput Netw Storage Anal"},{"key":"ref8","first-page":"89","article-title":"Optimizing stencil computations for NVIDIA Kepler GPUs","author":"maruyama","year":"2014","journal-title":"Proc Int Workshop High-Performance Stencil Comput"},{"key":"ref7","first-page":"41:1","article-title":"STELLA: A domain-specific tool for structured grid methods in weather and climate models","author":"gysi","year":"2015","journal-title":"Proc Int Conf High Performance Comput Netw Storage Anal"},{"doi-asserted-by":"publisher","key":"ref2","DOI":"10.1145\/2259016.2259037"},{"key":"ref9","first-page":"1","article-title":"3.5-D blocking optimization for stencil computations on modern CPUs and GPUs","author":"nguyen","year":"2010","journal-title":"Proc ACM\/IEEE Int Conf High Performance Comput Netw Storage Anal"},{"key":"ref1","first-page":"256","article-title":"Performance modeling and automatic ghost zone optimization for iterative stencil loops on GPUs","author":"meng","year":"2009","journal-title":"Proc 23rd Int'l Conf Supercomputing"},{"key":"ref20","first-page":"361","article-title":"A performance model and efficiency-based assignment of buffering strategies for automatic GPU stencil code generation","author":"hu","year":"2016","journal-title":"Proc IEEE 10th Int Symp Embedded Multicore\/Many-Core Syst -on-Chip"},{"key":"ref22","first-page":"1","article-title":"Model-driven auto-tuning of stencil computations on GPUs","author":"hu","year":"2015","journal-title":"Proc Int Workshop High-Performance Stencil Comput"},{"key":"ref21","first-page":"152","article-title":"An analytical model for a GPU architecture with memory-level and thread-level parallelism awareness","author":"hong","year":"2009","journal-title":"Proc 36th Annu Int Symp Comput Archit"},{"key":"ref24","first-page":"843","article-title":"Tight bounds on capacity misses for 3D stencil codes","author":"leopold","year":"2002","journal-title":"Proc Int Conf Comput Sci"},{"doi-asserted-by":"publisher","key":"ref23","DOI":"10.1137\/070693199"},{"key":"ref26","first-page":"149","article-title":"Performance and power analysis of ATI GPU: A statistical approach","author":"zhang","year":"2011","journal-title":"Proc 6th IEEE Int Conf Netw Archit Storage"},{"doi-asserted-by":"publisher","key":"ref25","DOI":"10.1007\/s11227-015-1392-1"}],"container-title":["IEEE Transactions on Multi-Scale Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielaam\/6687315\/8466693\/7930466-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6687315\/8466693\/07930466.pdf?arnumber=7930466","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,6,24]],"date-time":"2024-06-24T05:36:52Z","timestamp":1719207412000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/7930466\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,7,1]]},"references-count":40,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tmscs.2017.2705139","relation":{},"ISSN":["2332-7766","2372-207X"],"issn-type":[{"type":"electronic","value":"2332-7766"},{"type":"electronic","value":"2372-207X"}],"subject":[],"published":{"date-parts":[[2018,7,1]]}}}