{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T12:17:17Z","timestamp":1763468237437},"reference-count":27,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2015,2,26]],"date-time":"2015-02-26T00:00:00Z","timestamp":1424908800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2015,7]]},"DOI":"10.1007\/s11227-015-1392-1","type":"journal-article","created":{"date-parts":[[2015,2,25]],"date-time":"2015-02-25T06:01:11Z","timestamp":1424844071000},"page":"2433-2453","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["An analytical GPU performance model for 3D stencil computations from the angle of data traffic"],"prefix":"10.1007","volume":"71","author":[{"given":"Huayou","family":"Su","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xing","family":"Cai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mei","family":"Wen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chunyuan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,2,26]]},"reference":[{"key":"1392_CR1","doi-asserted-by":"crossref","unstructured":"Baghsorkhi SS, Delahaye M, Patel SJ, Gropp WD, Hwu WMW (2010) An adaptive performance modeling tool for GPU architectures. In: Proceedings of PPoPP\u201910. ACM, New York, pp 105\u2013114. doi: 10.1145\/1693453.1693470","DOI":"10.1145\/1693453.1693470"},{"key":"1392_CR2","doi-asserted-by":"crossref","unstructured":"Bakhoda A, Yuan GL, Fung WW, Wong H, Aamodt TM (2009) Analyzing cuda workloads using a detailed GPU simulator. In: IEEE international symposium on performance analysis of systems and software (ISPASS\u201909). IEEE, pp 163\u2013174","DOI":"10.1109\/ISPASS.2009.4919648"},{"key":"1392_CR3","doi-asserted-by":"crossref","unstructured":"Datta K, Murphy M, Volkov V, Williams S, Carter J, Oliker L, Patterson D, Shalf J, Yelick K (2008) Stencil computation optimization and auto-tuning on state-of-the-art multicore architectures. In: Proceedings of SC\u201908. IEEE Press, Piscataway, pp 4:1\u20134:12. doi: 10.1109\/SC.2008.5222004","DOI":"10.1109\/SC.2008.5222004"},{"issue":"1","key":"1392_CR4","doi-asserted-by":"crossref","first-page":"129","DOI":"10.1137\/070693199","volume":"51","author":"K Datta","year":"2009","unstructured":"Datta K, Kamil S, Williams S, Oliker L, Shalf J, Yelick K (2009) Optimization and performance modeling of stencil computations on modern microprocessors. SIAM Rev 51(1):129\u2013159","journal-title":"SIAM Rev"},{"key":"1392_CR5","unstructured":"de la Cruz R, Araya-Polo M (in press) Modeling stencil computations on modern HPC architectures"},{"issue":"3","key":"1392_CR6","first-page":"23","volume":"40","author":"R Cruz De La","year":"2014","unstructured":"De La Cruz R, Araya-Polo M (2014) Algorithm 942: semi-stencil. ACM Trans Math Softw (TOMS) 40(3):23","journal-title":"ACM Trans Math Softw (TOMS)"},{"key":"1392_CR7","doi-asserted-by":"crossref","unstructured":"Holewinski J, Pouchet LN, Sadayappan P (2012) High-performance code generation for stencil computations on GPU architectures. In: Proceedings of ICS\u201912. ACM, New York, pp 311\u2013320. doi: 10.1145\/2304576.2304619","DOI":"10.1145\/2304576.2304619"},{"key":"1392_CR8","doi-asserted-by":"crossref","unstructured":"Hong S, Kim H (2009) An analytical model for a GPU architecture with memory-level and thread-level parallelism awareness. In: Proceedings of ISCA\u201909. ACM, New York, pp 152\u2013163. doi: 10.1145\/1555754.1555775","DOI":"10.1145\/1555754.1555775"},{"key":"1392_CR9","doi-asserted-by":"crossref","unstructured":"Kamil S, Husbands P, Oliker L, Shalf J, Yelick K (2005) Impact of modern memory subsystems on cache optimizations for stencil computations. In: Proceedings of MSP\u201905. ACM, New York, pp 36\u201343. doi: 10.1145\/1111583.1111589","DOI":"10.1145\/1111583.1111589"},{"key":"1392_CR10","doi-asserted-by":"crossref","unstructured":"Kamil S, Datta K, Williams S, Oliker L, Shalf J, Yelick K (2006) Implicit and explicit optimizations for stencil computations. In: Proceedings of MSPC\u201906. ACM, New York, pp 51\u201360. doi: 10.1145\/1178597.1178605","DOI":"10.1145\/1178597.1178605"},{"key":"1392_CR11","doi-asserted-by":"crossref","unstructured":"Kamil S, Chan C, Oliker L, Shalf J, Williams S (2010) An auto-tuning framework for parallel multicore stencil computations. In: Proceedings of IPDPS\u201910, pp 1\u201312. doi: 10.1109\/IPDPS.2010.5470421","DOI":"10.1109\/IPDPS.2010.5470421"},{"key":"1392_CR12","doi-asserted-by":"crossref","unstructured":"Meng J, Skadron K (2009) Performance modeling and automatic ghost zone optimization for iterative stencil loops on GPUs. In: Proceedings of ICS\u201909. ACM, New York, pp 256\u2013265. doi: 10.1145\/1542275.1542313","DOI":"10.1145\/1542275.1542313"},{"key":"1392_CR13","doi-asserted-by":"crossref","unstructured":"Micikevicius P (2009) 3D finite difference computation on GPUs using CUDA. In: GPGPU-2. ACM, New York, pp 79\u201384. doi: 10.1145\/1513895.1513905","DOI":"10.1145\/1513895.1513905"},{"issue":"2","key":"1392_CR14","doi-asserted-by":"crossref","first-page":"56","DOI":"10.1109\/MM.2010.41","volume":"30","author":"J Nickolls","year":"2010","unstructured":"Nickolls J, Dally W (2010) The GPU computing era. Micro IEEE 30(2):56\u201369. doi: 10.1109\/MM.2010.41","journal-title":"Micro IEEE"},{"key":"1392_CR15","doi-asserted-by":"crossref","unstructured":"Nugteren C, van den Braak GJ, Corporaal H, Bal H (2014) A detailed GPU cache model based on reuse distance theory. In: IEEE 20th international symposium on high performance computer architecture (HPCA). IEEE, pp 37\u201348","DOI":"10.1109\/HPCA.2014.6835955"},{"key":"1392_CR16","unstructured":"NVIDIA T (2013) K20-k20x GPU accelerators benchmarks. ApplicationPerformance Technical Brief, Nvidia. http:\/\/www.nvidia.com\/docs\/IO\/122874\/K20-and-K20X-application-performance-technical-brief.pdf"},{"key":"1392_CR17","unstructured":"NVIDIA C (2012a) CUDA API reference manual"},{"key":"1392_CR18","unstructured":"Profiler user\u2019s guide. http:\/\/docs.nvidia.com\/cuda\/pdf\/CUDA_Profiler_Users_Guide.pdf"},{"key":"1392_CR19","doi-asserted-by":"crossref","unstructured":"Rahman SMF, Yi Q, Qasem A (2011) Understanding stencil code performance on multicore architectures. In: Proceedings of the 8th ACM international conference on computing frontiers. ACM, New York p 30","DOI":"10.1145\/2016604.2016641"},{"key":"1392_CR20","doi-asserted-by":"crossref","first-page":"2027","DOI":"10.1016\/j.procs.2011.04.221","volume":"4","author":"A Sch\u00e4fer","year":"2011","unstructured":"Sch\u00e4fer A, Fey D (2011) High performance stencil code algorithms for GPGPUs. Procedia Comput Sci 4:2027\u20132036","journal-title":"Procedia Comput Sci"},{"key":"1392_CR21","doi-asserted-by":"crossref","unstructured":"Sim J, Dasgupta A, Kim H, Vuduc R (2012) A performance analysis framework for identifying potential benefits in GPGPU applications. In: Proceedings of PPoPP\u201912. ACM, New York, pp 11\u201322. doi: 10.1145\/2145816.2145819","DOI":"10.1145\/2145816.2145819"},{"key":"1392_CR22","unstructured":"Stengel H, Treibig J, Hager G, Wellein G (2014) Quantifying performance bottlenecks of stencil computations using the execution-cache-memory model. arXiv:1410.5010"},{"key":"1392_CR23","doi-asserted-by":"crossref","unstructured":"Su H, Wu N, Wen M, Zhang C, Cai X (2013a) On the GPU\u2013CPU performance portability of OpenCL for 3D stencil computations. In: International conference on parallel and distributed systems (ICPADS). IEEE, pp 78\u201385","DOI":"10.1109\/ICPADS.2013.23"},{"key":"1392_CR24","doi-asserted-by":"crossref","unstructured":"Su H, Wu N, Wen M, Zhang C, Cai X (2013b) On the GPU performance of 3D stencil computations implemented in OpenCL. In: Supercomputing. Springer, New York, pp 125\u2013135","DOI":"10.1007\/978-3-642-38750-0_10"},{"key":"1392_CR25","doi-asserted-by":"crossref","unstructured":"Unat D, Cai X, Baden SB (2011) Mint: realizing CUDA performance in 3D stencil methods with annotated C. In: Proceedings of ICS\u201911. ACM, New York, pp 214\u2013224. doi: 10.1145\/1995896.1995932","DOI":"10.1145\/1995896.1995932"},{"key":"1392_CR26","doi-asserted-by":"crossref","unstructured":"Williams S, Waterman A, Patterson D (2009) Roofline: an insightful visual performance model for multicore architectures. Commun ACM 52(4):65\u201376. doi: 10.1145\/1498765.1498785","DOI":"10.1145\/1498765.1498785"},{"key":"1392_CR27","doi-asserted-by":"crossref","unstructured":"Zhang Y, Mueller F (2012) Auto-generation and auto-tuning of 3D stencil codes on GPU clusters. In: Proceedings of CGO\u201912. ACM, New York, pp 155\u2013164. doi: 10.1145\/2259016.2259037","DOI":"10.1145\/2259016.2259037"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-015-1392-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11227-015-1392-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-015-1392-1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,8,21]],"date-time":"2019-08-21T07:00:54Z","timestamp":1566370854000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11227-015-1392-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,2,26]]},"references-count":27,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2015,7]]}},"alternative-id":["1392"],"URL":"https:\/\/doi.org\/10.1007\/s11227-015-1392-1","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2015,2,26]]}}}