{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T15:15:39Z","timestamp":1743088539454,"version":"3.40.3"},"publisher-location":"Berlin, Heidelberg","reference-count":17,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642387494"},{"type":"electronic","value":"9783642387500"}],"license":[{"start":{"date-parts":[[2013,1,1]],"date-time":"2013-01-01T00:00:00Z","timestamp":1356998400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2013,1,1]],"date-time":"2013-01-01T00:00:00Z","timestamp":1356998400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2013]]},"DOI":"10.1007\/978-3-642-38750-0_10","type":"book-chapter","created":{"date-parts":[[2013,6,10]],"date-time":"2013-06-10T01:26:27Z","timestamp":1370827587000},"page":"125-135","source":"Crossref","is-referenced-by-count":8,"title":["On the GPU Performance of 3D Stencil Computations Implemented in OpenCL"],"prefix":"10.1007","author":[{"given":"Huayou","family":"Su","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nan","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mei","family":"Wen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chunyuan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xing","family":"Cai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"10_CR1","unstructured":"Khronos OpenCL Working Group: The OpenCL Specification (2011), \n                    http:\/\/www.khronos.org\/registry\/cl\/specs\/opencl-1.1.pdf"},{"key":"10_CR2","doi-asserted-by":"crossref","unstructured":"Fang, J., Varbanescu, A., Sips, H.: A comprehensive performance comparison of CUDA and OpenCL. In: Proceedings of the 2011 International Conference on Parallel Processing, pp. 216\u2013225. IEEE Computer Society Press (2011)","DOI":"10.1109\/ICPP.2011.45"},{"key":"10_CR3","unstructured":"Karimi, K., Dickson, N., Hamze, F.: A performance comparison of CUDA and OpenCL (2010), \n                    http:\/\/arxiv.org\/ftp\/arxiv\/papers\/1005\/1005.2581.pdf"},{"key":"10_CR4","unstructured":"Komatsu, K., Sato, K., Arai, Y., Koyama, K., Takizawa, H., Kobayashi, H.: Evaluating performance and portability of OpenCL programs. In: Proceedings of the Fifth International Workshop on Automatic Performance Tuning (iWAPT 2010). IEEE Computer Society Press (2010)"},{"issue":"8","key":"10_CR5","doi-asserted-by":"publisher","first-page":"391","DOI":"10.1016\/j.parco.2011.10.002","volume":"38","author":"P. Du","year":"2012","unstructured":"Du, P., Weber, R., Luszczek, P., Tomov, S., Peterson, G., Dongarra, J.: From CUDA to OpenCL: Towards a performance-portable solution for multi-platform GPU programming. Parallel Computing\u00a038(8), 391\u2013407 (2012)","journal-title":"Parallel Computing"},{"key":"10_CR6","doi-asserted-by":"crossref","unstructured":"Unat, D., Cai, X., Baden, S.: Mint: realizing CUDA performance in 3D stencil methods with annotated C. In: Proceedings of the 25th ACM International Conference on Supercomputing, pp. 214\u2013224. ACM (2011)","DOI":"10.1145\/1995896.1995932"},{"key":"10_CR7","doi-asserted-by":"crossref","unstructured":"Sch\u00e4fer, A., Fey, D.: High performance stencil code algorithms for GPGPUs. In: Proceedings of the International Conference on Computational Science. Procedia Computer Science, vol.\u00a04, pp. 2027\u20132036. Elsevier (2011)","DOI":"10.1016\/j.procs.2011.04.221"},{"key":"10_CR8","unstructured":"NVIDIA: NVIDIA OpenCL Best Practices Guide (2009), \n                    http:\/\/developer.download.nvidia.com\/compute\/cuda\/2_3\/opencl\/docs\/NVIDIA_OpenCL_BestPracticesGuide.pdf"},{"key":"10_CR9","unstructured":"NVIDIA: NVIDIA OpenCL SDK code sample of 3D FDTD, \n                    http:\/\/developer.download.nvidia.com\/compute\/DevZone\/OpenCL\/Projects\/oclFDTD3d.zip"},{"key":"10_CR10","doi-asserted-by":"crossref","unstructured":"Datta, K., Murphy, M., Volkov, V., Williams, S., Carter, J., Oliker, L., Patterson, D., Shalf, J., Yelick, K.: Stencil computation optimization and auto-tuning on state-of-the-art multicore architectures. In: Proceedings of the 2008 ACM\/IEEE Conference on Supercomputing. IEEE Computer Society Press (2008)","DOI":"10.1109\/SC.2008.5222004"},{"key":"10_CR11","doi-asserted-by":"crossref","unstructured":"Holewinski, J., Pouchet, L.N., Sadayappan, P.: High-performance code generation for stencil computations on GPU architectures. In: Proceedings of the 26th ACM International Conference on Supercomputing, pp. 311\u2013320. ACM (2012)","DOI":"10.1145\/2304576.2304619"},{"key":"10_CR12","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Mueller, F.: Auto-generation and auto-tuning of 3D stencil codes on GPU clusters. In: Proceedings of the Tenth International Symposium on Code Generation and Optimization, pp. 155\u2013164. ACM (2012)","DOI":"10.1145\/2259016.2259037"},{"key":"10_CR13","doi-asserted-by":"crossref","unstructured":"Nguyen, A., Satish, N., Chhugani, J., Kim, C., Dubey, P.: 3.5-D blocking optimization for stencil computations on modern CPUs and GPUs. In: Proceedings of the 2010 ACM\/IEEE International Conference for High Performance Computing, Networking, Storage and Analysis. IEEE Computer Society Press (2010)","DOI":"10.1109\/SC.2010.2"},{"key":"10_CR14","doi-asserted-by":"crossref","unstructured":"Zumbusch, G.: Tuning a finite difference computation for parallel vector processors. In: Proceedings of the 2012 11th International Symposium on Parallel and Distributed Computing, pp. 63\u201370. IEEE Computer Society Press (2012)","DOI":"10.1109\/ISPDC.2012.17"},{"issue":"1","key":"10_CR15","doi-asserted-by":"publisher","first-page":"57","DOI":"10.1007\/s11390-012-1206-3","volume":"27","author":"Y. Yang","year":"2012","unstructured":"Yang, Y., Cui, H., Feng, X., Xue, J.: A hybrid circular queue method for iterative stencil computations on GPUs. Journal of Computer Science and Technology\u00a027(1), 57\u201374 (2012)","journal-title":"Journal of Computer Science and Technology"},{"key":"10_CR16","unstructured":"Rul, S., Vandierendonck, H., D\u2019Haene, J., De Bosschere, K.: An experimental study on performance portability of OpenCL kernels. In: Symposium on Application Accelerators in High Performance Computing, SAAHPC 2010 (2010)"},{"key":"10_CR17","unstructured":"Demidov, D.: VexCL: Vector expression template library for OpenCL (2013), \n                    http:\/\/www.codeproject.com\/Articles\/415058\/VexCL-Vector-expression-template-library-for-OpenC"}],"container-title":["Lecture Notes in Computer Science","Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-38750-0_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,12,17]],"date-time":"2021-12-17T09:15:08Z","timestamp":1639732508000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-642-38750-0_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013]]},"ISBN":["9783642387494","9783642387500"],"references-count":17,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-38750-0_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2013]]}}}