{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T15:00:11Z","timestamp":1784905211756,"version":"3.55.0"},"publisher-location":"Cham","reference-count":28,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783319676296","type":"print"},{"value":"9783319676302","type":"electronic"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-67630-2_36","type":"book-chapter","created":{"date-parts":[[2017,10,19]],"date-time":"2017-10-19T04:33:17Z","timestamp":1508387597000},"page":"496-514","source":"Crossref","is-referenced-by-count":32,"title":["Tuning and Optimization for a Variety of Many-Core Architectures Without Changing a Single Line of Implementation Code Using the Alpaka Library"],"prefix":"10.1007","author":[{"given":"Alexander","family":"Matthes","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ren\u00e9","family":"Widera","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Erik","family":"Zenker","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Benjamin","family":"Worpitz","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Axel","family":"Huebl","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Michael","family":"Bussmann","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2017,10,20]]},"reference":[{"key":"36_CR1","unstructured":"AMD: HIP DATA SHEET - It\u2019s HIP to be Open, November 2015, https:\/\/gpuopen.com\/wp-content\/uploads\/2016\/01\/7637_HIP_Datasheet_V1_7_PrintReady_US_WE.pdf . Accessed 11 April 2017"},{"issue":"10","key":"36_CR2","doi-asserted-by":"crossref","first-page":"2831","DOI":"10.1109\/TPS.2010.2064310","volume":"38","author":"H Burau","year":"2010","unstructured":"Burau, H., Widera, R., Honig, W., Juckeland, G., Debus, A., Kluge, T., Schramm, U., Cowan, T.E., Sauerbrey, R., Bussmann, M.: Picongpu: a fully relativistic particle-in-cell code for a gpu cluster. IEEE Trans. Plasma Sci. 38(10), 2831\u20132839 (2010)","journal-title":"IEEE Trans. Plasma Sci."},{"key":"36_CR3","unstructured":"Bussmann, M., Burau, H., Cowan, T.E., Debus, A., Huebl, A., Juckeland, G., Kluge, T., Nagel, W.E., Pausch, R., Schmitt, F., Schramm, U., Schuchart, J., Widera, R.: Radiative signatures of the relativistic kelvin-helmholtz instability. In: Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis, SC 2013, NY, USA, pp. 5:1\u20135:12 (2013), http:\/\/doi.acm.org\/10.1145\/2503210.2504564"},{"issue":"1","key":"36_CR4","doi-asserted-by":"crossref","first-page":"46","DOI":"10.1109\/99.660313","volume":"5","author":"L Dagum","year":"1998","unstructured":"Dagum, L., Menon, R.: Openmp: an industry standard API for shared-memory programming. IEEE Comput. Sci. Eng. 5(1), 46\u201355 (1998)","journal-title":"IEEE Comput. Sci. Eng."},{"key":"36_CR5","doi-asserted-by":"crossref","first-page":"362","DOI":"10.1016\/j.cpc.2016.05.019","volume":"207","author":"C Eckert","year":"2016","unstructured":"Eckert, C., Zenker, E., Bussmann, M., Albach, D.: Haseongpu - an adaptive, load-balanced mpi\/gpu-code for calculating the amplified spontaneous emission in high power laser media. Comput. Phys. Commun. 207, 362\u2013374 (2016)","journal-title":"Comput. Phys. Commun."},{"key":"36_CR6","doi-asserted-by":"crossref","unstructured":"Edwards, H.C., Trott, C.R.: Kokkos: enabling performance portability across manycore architectures. In: 2013 Extreme Scaling Workshop (XSW 2013), pp. 18\u201324. IEEE (2013)","DOI":"10.1109\/XSW.2013.7"},{"key":"36_CR7","doi-asserted-by":"crossref","unstructured":"Khronos Group: The opencl specification - Version 2.1, 11 November 2015, https:\/\/www.khronos.org\/registry\/cl\/specs\/opencl-2.1.pdf . Accessed 23 March 2017","DOI":"10.1145\/2791321.2791337"},{"key":"36_CR8","unstructured":"Gumhold, S.: Lecture \u201cScientific Visualization\u201d (2011)"},{"key":"36_CR9","unstructured":"Hernandez, O.: Overview of the Power8 Architecture (2016), https:\/\/indico-jsc.fz-juelich.de\/event\/24\/session\/24\/contribution\/0\/material\/slides\/ . Accessed 24 March 2017"},{"key":"36_CR10","doi-asserted-by":"crossref","DOI":"10.2172\/1169830","volume-title":"The Raja Portability Layer: Overview and Status","author":"R Hornung","year":"2014","unstructured":"Hornung, R., Keasler, J., et al.: The Raja Portability Layer: Overview and Status. Lawrence Livermore National Laboratory, Livermore (2014)"},{"key":"36_CR11","unstructured":"Intel Corporation: Intel Threading Building Blocks, https:\/\/www.threadingbuildingblocks.org\/ . Accessed 12 April 2017"},{"key":"36_CR12","doi-asserted-by":"crossref","unstructured":"Jeffers, J., Reinders, J., Sodani, A.: Intel Xeon Phi Processor High Performance Programming Knights Landing Edition. Morgan Kaufmann, 1 July 2016","DOI":"10.1016\/B978-0-12-809194-4.09999-3"},{"key":"36_CR13","unstructured":"Khronos OpenCL Working Group SYCL subgroup: Sycl specification - Version 1.2., 8 May 2015, https:\/\/www.khronos.org\/registry\/sycl\/specs\/sycl-1.2.pdf . Accessed 23 March 2017"},{"key":"36_CR14","doi-asserted-by":"crossref","unstructured":"Li, J., Li, X., Tan, G., Chen, M., Sun, N.: An optimized large-scale hybrid DGEMM design for CPUs and ATI GPUs. In: Proceedings of the 26th ACM International Conference on Supercomputing, pp. 377\u2013386. ACM (2012)","DOI":"10.1145\/2304576.2304626"},{"key":"36_CR15","unstructured":"Matthes, A., Widera, R., Zenker, E., Worpitz, B., H\u00fcbl, A., Bussmann, M.: Matrix multiplication software and results bundle for paper Tuning and optimization for a variety of many-core architectures without changing a single line of implementation code using the Alpaka library for P $$^{\\wedge }$$ 3MA submission, April 2017, https:\/\/doi.org\/10.5281\/zenodo.439528"},{"key":"36_CR16","unstructured":"Meuer, H.W., Strohmaier, E., Dongarra, J., Simon, H., Meuer, M.: November 2016 \u2014 TOP500 Supercomputer Sites, November 2016"},{"key":"36_CR17","unstructured":"Microsoft Corporation: C++ amp : language and programming model - Version 1.2, December 2013, http:\/\/download.microsoft.com\/download\/2\/2\/9\/22972859-15c2-4d96-97ae-93344241d56c\/cppampopenspecificationv12.pdf . Accessed 23 March 2017"},{"key":"36_CR18","unstructured":"Newman, B.: Intel Xeon E5\u20132600 v3 \u201cHaswell\u201d Processor Review \u2014 Microway, 8 September 2014, https:\/\/www.microway.com\/hpc-tech-tips\/intel-xeon-e5-2600-v3-haswell-processor-review\/ . Accessed 24 March 2017"},{"key":"36_CR19","unstructured":"Nvidia: Tesla K80 HPC and Machine Learning Accelerator (2014), https:\/\/www.nvidia.com\/object\/tesla-k80.html . Accessed 23 March 2017"},{"key":"36_CR20","unstructured":"Nvidia: Tesla P100 Most Advanced Data Center Accelerator (2016), https:\/\/www.nvidia.com\/object\/tesla-p100.html . Accessed 23 March 2017"},{"key":"36_CR21","unstructured":"Nvidia Corporation: NVIDIAs Next Generation - CUDA Compute Architecture: Kepler GK110\/210. Whitepaper (2014)"},{"key":"36_CR22","unstructured":"Nvidia Corporation: NVIDIA Tesla P100 - The Most Advanced Datacenter Accelerator Ever Built. WP-08019-001_v01.1., May 2016"},{"key":"36_CR23","unstructured":"Nvidia Corporation: NVIDIA CUDA C Programming Guide Version 8.0., January 2017, http:\/\/docs.nvidia.com\/cuda\/pdf\/CUDA_C_Programming_Guide.pdf . Accessed 23 March 2017"},{"key":"36_CR24","unstructured":"OpenACC-Standard.org: The OpenACC Application Programming Interface - Version 2.5, October 2015, http:\/\/www.openacc.org\/sites\/default\/files\/OpenACC_2pt5.pdf , Accessed 23 March 2017"},{"key":"36_CR25","unstructured":"Wong, M., Andrew, R., Rovatsou, M., Reyes, R.: Khronos\u2019s OpenCL SYCL to support Heterogeneous Devices for C++, 12 February 2016, http:\/\/www.open-std.org\/jtc1\/sc22\/wg21\/docs\/papers\/2016\/p0236r0.pdf . Accessed 23 March 2017"},{"key":"36_CR26","unstructured":"Zenker, E.: Graybat - Graph Approach for Highly Generic Communication Schemes Based on Adaptive Topologies, 5 March 2016, https:\/\/github.com\/ComputationalRadiationPhysics\/graybat"},{"key":"36_CR27","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1007\/978-3-319-46079-6_21","volume-title":"High Performance Computing","author":"E Zenker","year":"2016","unstructured":"Zenker, E., Widera, R., Huebl, A., Juckeland, G., Kn\u00fcpfer, A., Nagel, W.E., Bussmann, M.: Performance-portable many-core plasma simulations: porting PIConGPU to OpenPower and beyond. In: Taufer, M., Mohr, B., Kunkel, J.M. (eds.) ISC High Performance 2016. LNCS, vol. 9945, pp. 293\u2013301. Springer, Cham (2016). doi: 10.1007\/978-3-319-46079-6_21"},{"key":"36_CR28","doi-asserted-by":"crossref","unstructured":"Zenker, E., Worpitz, B., Widera, R., Huebl, A., Juckeland, G., Kn\u00fcpfer, A., Nagel, W.E., Bussmann, M.: Alpaka-an abstraction library for parallel kernel acceleration. In: 2016 IEEE International Parallel and Distributed Processing Symposium Workshops, pp. 631\u2013640. IEEE (2016)","DOI":"10.1109\/IPDPSW.2016.50"}],"container-title":["Lecture Notes in Computer Science","High Performance Computing"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-67630-2_36","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,26]],"date-time":"2025-06-26T15:25:51Z","timestamp":1750951551000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-67630-2_36"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319676296","9783319676302"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-67630-2_36","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2017]]}}}