{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T09:49:47Z","timestamp":1764841787914,"version":"3.40.3"},"publisher-location":"Cham","reference-count":25,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319174723"},{"type":"electronic","value":"9783319174730"}],"license":[{"start":{"date-parts":[[2015,1,1]],"date-time":"2015-01-01T00:00:00Z","timestamp":1420070400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2015,1,1]],"date-time":"2015-01-01T00:00:00Z","timestamp":1420070400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2015]]},"DOI":"10.1007\/978-3-319-17473-0_5","type":"book-chapter","created":{"date-parts":[[2015,4,30]],"date-time":"2015-04-30T09:59:39Z","timestamp":1430387979000},"page":"67-81","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["NAS Parallel Benchmarks for GPGPUs Using a Directive-Based Programming Model"],"prefix":"10.1007","author":[{"given":"Rengan","family":"Xu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaonan","family":"Tian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sunita","family":"Chandrasekaran","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yonghong","family":"Yan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Barbara","family":"Chapman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,5,1]]},"reference":[{"key":"5_CR1","unstructured":"NPB-CUDA (2013). http:\/\/www.tu-chemnitz.de\/informatik\/PI\/forschung\/download\/npb-gpu\/"},{"key":"5_CR2","unstructured":"NPB-UPC (2013). http:\/\/threads.hpcl.gwu.edu\/sites\/npb-upc"},{"key":"5_CR3","unstructured":"OpenACC (2013). http:\/\/www.openacc-standard.org"},{"key":"5_CR4","unstructured":"OpenCL Standard (2013). http:\/\/www.khronos.org\/opencl"},{"key":"5_CR5","unstructured":"OpenMP (2013). www.openmp.org"},{"key":"5_CR6","unstructured":"11 Tricks for Maximizing Performance with OpenACC Directives in Fortran (2014). http:\/\/www.pgroup.com\/resources\/openacc_tips_fortran.htm"},{"key":"5_CR7","unstructured":"CUDA (2014). http:\/\/www.nvidia.com\/object\/cuda_home_new.html"},{"key":"5_CR8","unstructured":"CUDA C Programming Guide (2014). http:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/"},{"key":"5_CR9","unstructured":"Pathscale NPB2.3 OpenACC (2014). https:\/\/github.com\/pathscale\/NPB2.3-OpenACC-C"},{"key":"5_CR10","unstructured":"Bailey, D., et al.: The NAS Parallel Benchmarks. NASA Ames Research Center (1994)"},{"key":"5_CR11","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"crossref","first-page":"74","DOI":"10.1007\/978-3-319-05215-1_6","volume-title":"OpenSHMEM and Related Technologies","author":"M Baker","year":"2014","unstructured":"Baker, M., Pophale, S., Vasnier, J.-C., Jin, H., Hernandez, O.: Hybrid programming using OpenSHMEM and OpenACC. In: Poole, S., Hernandez, O., Shamis, P. (eds.) OpenSHMEM 2014. LNCS, vol. 8356, pp. 74\u201389. Springer, Heidelberg (2014)"},{"key":"5_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"13","DOI":"10.1007\/978-3-642-35893-7_2","volume-title":"Facing the Multicore-Challenge III","author":"W Ding","year":"2013","unstructured":"Ding, W., Hernandez, O., Chapman, B.: A similarity-based analysis tool for porting OpenMP applications. In: Keller, R., Kramer, D., Weiss, J.-P. (eds.) Facing the Multicore-Challenge III. LNCS, vol. 7686, pp. 13\u201324. Springer, Heidelberg (2013)"},{"issue":"8","key":"5_CR13","doi-asserted-by":"publisher","first-page":"1072","DOI":"10.1002\/cpe.2903","volume":"25","author":"W Ding","year":"2013","unstructured":"Ding, W., Hsu, C.-H., Hernandez, O., Chapman, B.M., Graham, R.L.: KLONOS: similarity-based planning tool support for porting scientific applications. Concurrency Comput. Pract. Experience 25(8), 1072\u20131088 (2013)","journal-title":"Concurrency Comput. Pract. Experience"},{"key":"5_CR14","unstructured":"Dolbeau, R., Bihan, S., Bodin, F.: HMPP: a hybrid multi-core parallel programming environment. In: Workshop on GPGPU (2007)"},{"key":"5_CR15","unstructured":"Frumkin, M., Jin, H., Yan, J.: Implementation of NAS parallel benchmarks in high performance fortran. NAS Techinical report NAS-98-009 (1998)"},{"key":"5_CR16","doi-asserted-by":"crossref","unstructured":"Grewe, D., Wang, Z., O\u2019Boyle, M.F.: Portable mapping of data parallel programs to OpenCL for heterogeneous systems. In: 2013 IEEE\/ACM International Symposium on CGO, pp. 1\u201310. IEEE (2013)","DOI":"10.1109\/CGO.2013.6494993"},{"issue":"39","key":"5_CR17","first-page":"851","volume":"3","author":"M Harris","year":"2007","unstructured":"Harris, M., Sengupta, S., Owens, J.D.: Parallel prefix sum (scan) with CUDA. GPU Gems 3(39), 851\u2013876 (2007)","journal-title":"GPU Gems"},{"key":"5_CR18","unstructured":"Jin, H., Frumkin, M., Yan, J.: The OpenMP implementation of NAS parallel benchmarks and its performance. Technical report, NAS-99-011, NASA Ames Research Center (1999)"},{"key":"5_CR19","doi-asserted-by":"crossref","unstructured":"Lee, S., Li, D., Vetter, J.S.: Interactive program debugging and optimization for directive-based, Efficient GPU Computing (2014)","DOI":"10.1109\/IPDPS.2014.57"},{"key":"5_CR20","doi-asserted-by":"crossref","unstructured":"Lee, S., Vetter, J.S.: Early evaluation of directive-based GPU programming models for productive exascale computing. In: SC 2012, pp. 23:1\u201323:11. IEEE Computer Society Press (2012)","DOI":"10.1109\/SC.2012.51"},{"issue":"4","key":"5_CR21","doi-asserted-by":"publisher","first-page":"23","DOI":"10.1145\/1964218.1964223","volume":"38","author":"SJ Pennycook","year":"2011","unstructured":"Pennycook, S.J., Hammond, S.D., Jarvis, S.A., Mudalige, G.R.: Performance analysis of a hybrid MPI\/CUDA implementation of the NAS LU benchmark. ACM SIGMETRICS Perform. Eval. Rev. 38(4), 23\u201329 (2011)","journal-title":"ACM SIGMETRICS Perform. Eval. Rev."},{"key":"5_CR22","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"871","DOI":"10.1007\/978-3-642-32820-6_86","volume-title":"Euro-Par 2012 Parallel Processing","author":"R Reyes","year":"2012","unstructured":"Reyes, R., L\u00f3pez-Rodr\u00edguez, I., Fumero, J.J., de Sande, F.: accULL: an OpenACC implementation with CUDA and OpenCL support. In: Kaklamanis, C., Papatheodorou, T., Spirakis, P.G. (eds.) Euro-Par 2012. LNCS, vol. 7484, pp. 871\u2013882. Springer, Heidelberg (2012)"},{"key":"5_CR23","doi-asserted-by":"crossref","unstructured":"Seo, S., Jo, G., Lee, J.: Performance characterization of the NAS parallel benchmarks in OpenCL. In: IEEE International Symposium on IISWC, pp. 137\u2013148. IEEE (2011)","DOI":"10.1109\/IISWC.2011.6114174"},{"key":"5_CR24","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"105","DOI":"10.1007\/978-3-319-09967-5_6","volume-title":"Languages and Compilers for Parallel Computing - Testing","author":"X Tian","year":"2014","unstructured":"Tian, X., Xu, R., Yan, Y., Yun, Z., Chandrasekaran, S., Chapman, B.: Compiling a high-level directive-based programming model for GPGPUs. In: Ca\u1e63caval, C., Montesinos-Ortego, P. (eds.) LCPC 2013 - Testing. LNCS, vol. 8664, pp. 105\u2013120. Springer, Heidelberg (2014)"},{"issue":"2","key":"5_CR25","doi-asserted-by":"publisher","first-page":"154","DOI":"10.1093\/comjnl\/bxr063","volume":"55","author":"X Wu","year":"2012","unstructured":"Wu, X., Taylor, V.: Performance characteristics of hybrid MPI\/OpenMP implementations of NAS parallel benchmarks SP and BT on large-scale multicore clusters. Comput. J. 55(2), 154\u2013167 (2012)","journal-title":"Comput. J."}],"container-title":["Lecture Notes in Computer Science","Languages and Compilers for Parallel Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-17473-0_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,2,8]],"date-time":"2023-02-08T10:17:17Z","timestamp":1675851437000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-319-17473-0_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015]]},"ISBN":["9783319174723","9783319174730"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-17473-0_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2015]]},"assertion":[{"value":"1 May 2015","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}