{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,18]],"date-time":"2025-12-18T14:14:05Z","timestamp":1766067245888},"reference-count":38,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"12","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2020,12,1]]},"DOI":"10.1587\/transinf.2020pap0014","type":"journal-article","created":{"date-parts":[[2020,11,30]],"date-time":"2020-11-30T22:20:46Z","timestamp":1606774846000},"page":"2421-2434","source":"Crossref","is-referenced-by-count":4,"title":["A Data-Centric Directive-Based Framework to Accelerate Out-of-Core Stencil Computation on a GPU"],"prefix":"10.1587","volume":"E103.D","author":[{"given":"Jingcheng","family":"SHEN","sequence":"first","affiliation":[{"name":"Graduate School of Information Science and Technology, Osaka University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fumihiko","family":"INO","sequence":"additional","affiliation":[{"name":"Graduate School of Information Science and Technology, Osaka University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Albert","family":"FARR\u00c9S","sequence":"additional","affiliation":[{"name":"Barcelona Supercomputing Center"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mauricio","family":"HANZICH","sequence":"additional","affiliation":[{"name":"Barcelona Supercomputing Center"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"[1] M. Serpa, E. Cruz, M. Diener, A. Krause, A. Farr\u00e9s, C. Rosas, J. Panetta, M. Hanzich, and P. Navaux, \u201cStrategies to Improve the Performance of a Geophysics Model for Different Manycore Systems,\u201d Proc. 2017 International Symposium on Computer Architecture and High Performance Computing Workshops (SBAC-PADW), pp.49-54, Campinas, 2017. 10.1109\/sbac-padw.2017.17","DOI":"10.1109\/SBAC-PADW.2017.17"},{"key":"2","doi-asserted-by":"crossref","unstructured":"[2] A. Farres, C. Rosas, M. Hanzich, M. Jord\u00e0, and A. Pe\u00f1a, \u201cPerformance Evaluation of Fully Anisotropic Elastic Wave Propagation on NVIDIA Volta GPUs,\u201d Proc. 81st EAGE Conference and Exhibition, 2019. 10.3997\/2214-4609.201901307","DOI":"10.3997\/2214-4609.201901307"},{"key":"3","doi-asserted-by":"crossref","unstructured":"[3] S. Adams, J. Payne, and R. Boppana, \u201cFinite Difference Time Domain (FTDT) Simulations Using Graphics Processors,\u201d Proc. High Performance Computing Modernization Program Users Group Conf. (HPCMP-UGC), pp.334-338, Pittsburgh, PA, 2007. 10.1109\/hpcmp-ugc.2007.34","DOI":"10.1109\/HPCMP-UGC.2007.34"},{"key":"4","doi-asserted-by":"publisher","unstructured":"[4] K. Ikeda, F. Ino, and K. Hagihara, \u201cEfficient Acceleration of Mutual Information Computation for Nonrigid Registration using CUDA,\u201d IEEE J. Biomed. Health Inform., vol.18, no.3, pp.956-968, 2014. 10.1109\/jbhi.2014.2310745","DOI":"10.1109\/JBHI.2014.2310745"},{"key":"5","doi-asserted-by":"publisher","unstructured":"[5] S. Tabik, M. Peemen, and L. Romero, \u201cA tuning approach for iterative multiple 3d stencil pipeline on GPUs: Anisotropic Nonlinear Diffusion algorithm as case study,\u201d The Journal of Supercomputing, vol.74, no.4, pp.1580-1608, 2018. 10.1007\/s11227-017-2184-6","DOI":"10.1007\/s11227-017-2184-6"},{"key":"6","unstructured":"[6] K. Datta, \u201cAuto-tuning Stencil Codes for Cache-Based Multicore Platforms,\u201d Technical Report No. UCB\/EECS-2009-177, 2009."},{"key":"7","unstructured":"[7] A. Sch\u00e4fer and D. Fey, \u201cHigh Performance Stencil Code Algorithms for GPGPUs,\u201d Proc. International Conference of Computer Science, pp.2027-2036, 2011."},{"key":"8","doi-asserted-by":"publisher","unstructured":"[8] T. Okuyama, M. Okita, T. Abe, Y. Asai, H. Kitano, T. Nomura, and K. Hagihara, \u201cAccelerating ODE-based Simulation of General and Heterogeneous Biophysical Models using a GPU,\u201d IEEE Trans. Parallel Distrib. Syst., vol.25, no.8, pp.1966-1975, 2014. 10.1109\/tpds.2013.198","DOI":"10.1109\/TPDS.2013.198"},{"key":"9","doi-asserted-by":"publisher","unstructured":"[9] F. Ino, K. Shigeoka, T. Okuyama, M. Motokubota, and K. Hagihara, \u201cA Parallel Scheme for Accelerating Parameter Sweep Applications on a GPU,\u201d Concurrency and Computation: Practice and Experience, vol.26, no.2, pp.516-531, 2014. 10.1002\/cpe.3016","DOI":"10.1002\/cpe.3016"},{"key":"10","doi-asserted-by":"publisher","unstructured":"[10] Y. Mitani, F. Ino, and K. Hagihara, \u201cParallelizing Exact and Approximate String Matching via Inclusive Scan on a GPU,\u201d IEEE Trans. Parallel Distrib. Syst., vol.28, no.7, pp.1989-2002, 2017. 10.1109\/tpds.2016.2645222","DOI":"10.1109\/TPDS.2016.2645222"},{"key":"11","doi-asserted-by":"publisher","unstructured":"[11] J. Shen, K. Shigeoka, F. Ino, and K. Hagihara, \u201cGPU-based Branch-and-Bound Method to Solve Large 0-1 Knapsack Problems with Data-centric Strategies,\u201d Concurrency and Computation: Practice and Experience, vol.31, no.4, e4954, 2019. 10.1002\/cpe.4954","DOI":"10.1002\/cpe.4954"},{"key":"12","doi-asserted-by":"crossref","unstructured":"[12] W. Gropp, E. Lusk, and A. Skjellum, Using MPI: portable parallel programming with the message-passing interface, MIT press, 1999. 10.7551\/mitpress\/7056.001.0001","DOI":"10.7551\/mitpress\/7056.001.0001"},{"key":"13","unstructured":"[13] R. Pas, E. Stotzer, and C. Terboven, Using OpenMP-The Next Step: Affinity, Accelerators, Tasking, and SIMD, MIT press, 2017. 10.7551\/mitpress\/10031.001.0001"},{"key":"14","unstructured":"[14] NVIDIA Corporation, CUDA C Programming Guide, 2019. https:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/index.html."},{"key":"15","doi-asserted-by":"publisher","unstructured":"[15] Y. Lu, F. Ino, and K. Hagihara, \u201cCache-Aware GPU Optimization for Out-of-Core Cone Beam CT Reconstruction of High-Resolution Volumes,\u201d IEICE Trans. Inf. &amp; Syst., vol.E99-D, no.12, pp.3060-3071, 2016. 10.1587\/transinf.2016edp7174","DOI":"10.1587\/transinf.2016EDP7174"},{"key":"16","unstructured":"[16] PGI Compilers &amp; Tools, OpenACC Getting Started Guide, 2019. https:\/\/www.pgroup.com\/resources\/docs\/19.1\/pdf\/openacc19_gs.pdf."},{"key":"17","doi-asserted-by":"publisher","unstructured":"[17] N. Miki, F. Ino, and K. Hagihara, \u201cPACC: a directive-based programming framework for out-of-core stencil computation on accelerators,\u201d International Journal of High Performance Computing and Networking, vol.13, no.1, pp.19-34, 2019. 10.1504\/ijhpcn.2019.097046","DOI":"10.1504\/IJHPCN.2019.097046"},{"key":"18","doi-asserted-by":"publisher","unstructured":"[18] M. Sourouri, S. Baden, and X. Cai, \u201cPanda: A Compiler Framework for Concurrent CPU+GPU Execution of 3D Stencil Computations on GPU-accelerated Supercomputers,\u201d International Journal of Parallel Programming, vol.45, no.3, pp.711-729, 2017. 10.1007\/s10766-016-0454-1","DOI":"10.1007\/s10766-016-0454-1"},{"key":"19","doi-asserted-by":"crossref","unstructured":"[19] G. Jin, T. Endo, and S. Matsuoka, \u201cA multi-level optimization method for stencil computation on the domain that is bigger than memory capacity of GPU,\u201d Proc. 2013 IEEE International Symposium on Parallel Distributed Processing (IPDPS), Workshops and Phd Forum, pp.1080-1087, 2013. 10.1109\/ipdpsw.2013.58","DOI":"10.1109\/IPDPSW.2013.58"},{"key":"20","doi-asserted-by":"crossref","unstructured":"[20] T. Shimokawabe, T. Endo, N. Onodera, and T. Aoki, \u201cA stencil framework to realize large-scale computations beyond device memory capacity on GPU supercomputers,\u201d Proc. 2017 IEEE International Conference on Cluster Computing (CLUSTER), pp.525-529, Hawaii, USA, 2017. 10.1109\/cluster.2017.97","DOI":"10.1109\/CLUSTER.2017.97"},{"key":"21","doi-asserted-by":"crossref","unstructured":"[21] I. Reguly, G. Mudalige, and M. Giles, \u201cBeyond 16GB: out-of-core stencil computations,\u201d Proc. Workshop on Memory Centric Programming for HPC (MCHPC), pp.20-29, Denver, CO, 2017. 10.1145\/3145617.3145619","DOI":"10.1145\/3145617.3145619"},{"key":"22","doi-asserted-by":"crossref","unstructured":"[22] I. Reguly, G. Mudalige, M. Giles, D. Curran, and S. McIntosh-Smith, \u201cThe OPS domain specific abstraction for multi-block structured grid computations,\u201d Proc. 4th International Workshop on Domain-Specific Languages and High-Level Frameworks for High Performance Computing (WOLFHPC), pp.58-67, New Orleans, LA, 2014. 10.1109\/wolfhpc.2014.7","DOI":"10.1109\/WOLFHPC.2014.7"},{"key":"23","doi-asserted-by":"publisher","unstructured":"[23] I. Reguly, G. Mudalige, and M. Giles, \u201cLoop tiling in large-scale stencil codes at run-time with OPS,\u201d IEEE Trans. Parallel Distrib. Syst., vol.29, no.4, pp.873-886, 2017. 10.1109\/tpds.2017.2778161","DOI":"10.1109\/TPDS.2017.2778161"},{"key":"24","doi-asserted-by":"publisher","unstructured":"[24] G. Mudalige, I. Reguly, S. Jammy, C. Jacobs, M. Giles, and N. Sandham, \u201cLarge-scale performance of a DSL-based multi-block structured-mesh application for Direct Numerical Simulation,\u201d Journal of Parallel and Distributed Computing, vol.131, pp.130-146, 2019. 10.1016\/j.jpdc.2019.04.019","DOI":"10.1016\/j.jpdc.2019.04.019"},{"key":"25","doi-asserted-by":"crossref","unstructured":"[25] K. Hou, H. Wang, and W. Feng, \u201cGpu-unicache: Automatic code generation of spatial blocking for stencils on gpus,\u201d Proc. Computing Frontiers Conference (CF), pp.107-116, Siena, Italy, 2017. 10.1145\/3075564.3075583","DOI":"10.1145\/3075564.3075583"},{"key":"26","doi-asserted-by":"crossref","unstructured":"[26] T. Endo, \u201cApplying Recursive Temporal Blocking for Stencil Computations to Deeper Memory Hierarchy,\u201d Proc. IEEE 7th Non-Volatile Memory Systems and Applications Symposium (NVMSA), pp.19-24, Hakodate, Japan, 2018. 10.1109\/nvmsa.2018.00016","DOI":"10.1109\/NVMSA.2018.00016"},{"key":"27","doi-asserted-by":"publisher","unstructured":"[27] B. Fornberg, \u201cGeneration of finite difference formulas on arbitrarily spaced grids,\u201d Mathematics of computation, vol.51, no.184, pp.699-706, 1988. 10.1090\/s0025-5718-1988-0935077-0","DOI":"10.1090\/S0025-5718-1988-0935077-0"},{"key":"28","unstructured":"[28] S. Deldon, J. Beyer, and D. Miles, \u201cOpenACC and CUDA Unified Memory,\u201d Proc. Cray User Group (CUG), Stockholm, Sweden, 2018. https:\/\/cug.org\/proceedings\/cug2018_proceedings\/includes\/files\/pap115s2-file1.pdf."},{"key":"29","unstructured":"[29] N. Sakharnykh, \u201cMaximizing Unified Memory Performance in CUDA,\u201d 2017. https:\/\/devblogs.nvidia.com\/maximizing-unified-memory-performance-cuda\/."},{"key":"30","unstructured":"[30] J. McCalpin and D. Wonnacott, \u201cTime skewing: A value-based approach to optimizing for memory locality,\u201d Technical Report DCS-TR-379, Department of Computer Science, Rugers University, 1999."},{"key":"31","unstructured":"[31] D. Wonnacott, \u201cUsing time skewing to eliminate idle time due to memory bandwidth and network limitations,\u201d Proc. 14th International Parallel and Distributed Processing Symposium (IPDPS), pp.171-180, Cancun, Mexico, 2000. 10.1109\/ipdps.2000.845979"},{"key":"32","doi-asserted-by":"publisher","unstructured":"[32] T. Muranushi and J. Makino, \u201cOptimal temporal blocking for stencil computation,\u201d Procedia Computer Science, vol.51, pp.1303-1312, 2015. 10.1016\/j.procs.2015.05.315","DOI":"10.1016\/j.procs.2015.05.315"},{"key":"33","doi-asserted-by":"crossref","unstructured":"[33] T. Grosser, A. Cohen, P. Kelly, J. Ramanujam, P. Sadayappan, and S. Verdoolaege, \u201cSplit tiling for GPUs: automatic parallelization using trapezoidal tiles,\u201d Proc. 6th Workshop on General Purpose Processor Using Graphics Processing Units (GPGPU), pp.24-31, Houston, TX, 2013. 10.1145\/2458523.2458526","DOI":"10.1145\/2458523.2458526"},{"key":"34","doi-asserted-by":"publisher","unstructured":"[34] U. Bondhugula, V. Bandishti, and I. Pananilath, \u201cDiamond tiling: Tiling techniques to maximize parallelism for stencil computations,\u201d IEEE Trans. Parallel Distrib. Syst., vol.28, no.5, pp.1285-1298, 2016. 10.1109\/tpds.2016.2615094","DOI":"10.1109\/TPDS.2016.2615094"},{"key":"35","unstructured":"[35] M. Harris, \u201cHow to Optimize Data Transfers in CUDA C\/C++.\u201d https:\/\/devblogs.nvidia.com\/how-optimize-data-transfers-cuda-cc\/."},{"key":"36","doi-asserted-by":"crossref","unstructured":"[36] V. Allada, T. Benjegerdes, and B. Bode, \u201cPerformance analysis of memory transfers and GEMM subroutines on NVIDIA Tesla GPU cluster,\u201d Proc. 2009 IEEE International Conference on Cluster Computing and Workshops, pp.1-9, 2009. 10.1109\/clustr.2009.5289124","DOI":"10.1109\/CLUSTR.2009.5289124"},{"key":"37","doi-asserted-by":"crossref","unstructured":"[37] L. Wang, S. Chen, Y. Tang, and J. Su, \u201cGregex: GPU Based High Speed Regular Expression Matching Engine,\u201d Proc. 2011 5th International Conference on Innovative Mobile and Internet Services in Ubiquitous Computing (IMIS), pp.366-370, 2011. 10.1109\/imis.2011.107","DOI":"10.1109\/IMIS.2011.107"},{"key":"38","unstructured":"[38] R. Himeno, Himeno benchmark, 2015. http:\/\/accc.riken.jp\/en\/supercom\/himenobmt\/."}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E103.D\/12\/E103.D_2020PAP0014\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,12,5]],"date-time":"2020-12-05T04:39:36Z","timestamp":1607143176000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E103.D\/12\/E103.D_2020PAP0014\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,12,1]]},"references-count":38,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2020]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2020pap0014","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"value":"0916-8532","type":"print"},{"value":"1745-1361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,12,1]]}}}