{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,12]],"date-time":"2026-05-12T13:47:04Z","timestamp":1778593624592,"version":"3.51.4"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2022,12,6]],"date-time":"2022-12-06T00:00:00Z","timestamp":1670284800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"},{"start":{"date-parts":[[2022,12,6]],"date-time":"2022-12-06T00:00:00Z","timestamp":1670284800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"funder":[{"name":"ELLIIT","award":["GPAI"],"award-info":[{"award-number":["GPAI"]}]},{"name":"ELLIIT","award":["GPAI"],"award-info":[{"award-number":["GPAI"]}]},{"name":"SNIC","award":["2021\/22-971; LiU-gpu-2021-1"],"award-info":[{"award-number":["2021\/22-971; LiU-gpu-2021-1"]}]},{"name":"SNIC","award":["2021\/22-971; LiU-gpu-2021-1"],"award-info":[{"award-number":["2021\/22-971; LiU-gpu-2021-1"]}]},{"DOI":"10.13039\/501100003945","name":"Link\u00f6ping University","doi-asserted-by":"crossref","id":[{"id":"10.13039\/501100003945","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Parallel Prog"],"published-print":{"date-parts":[[2023,2]]},"abstract":"<jats:title>Abstract<\/jats:title><jats:p>We analyze the performance portability of the skeleton-based, single-source multi-backend high-level programming framework SkePU across multiple different CPU\u2013GPU heterogeneous systems. Thereby, we provide a systematic application efficiency characterization of SkePU-generated code in comparison to equivalent hand-written code in more low-level parallel programming models such as OpenMP and CUDA. For this purpose, we contribute ports of the STREAM benchmark suite and of a part of the NAS Parallel Benchmark suite to SkePU. We show that for STREAM and the EP benchmark, SkePU regularly scores efficiency values above 80% and in particular for CPU systems, SkePU can outperform hand-written code.<\/jats:p>","DOI":"10.1007\/s10766-022-00746-1","type":"journal-article","created":{"date-parts":[[2022,12,6]],"date-time":"2022-12-06T17:12:32Z","timestamp":1670346752000},"page":"61-82","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["Assessing Application Efficiency and Performance Portability in Single-Source Programming for Heterogeneous Parallel Systems"],"prefix":"10.1007","volume":"51","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6514-4601","authenticated-orcid":false,"given":"August","family":"Ernstsson","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4690-3964","authenticated-orcid":false,"given":"Dalvan","family":"Griebler","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5241-0026","authenticated-orcid":false,"given":"Christoph","family":"Kessler","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,12,6]]},"reference":[{"key":"746_CR1","doi-asserted-by":"crossref","unstructured":"Aldinucci, M., Danelutto, M., Kilpatrick, P., Torquati, M.: Fastflow: High-Level and Efficient Streaming on Multicore, chapter\u00a013, pp. 261\u2013280. Wiley, Hoboken (2017)","DOI":"10.1002\/9781119332015.ch13"},{"key":"746_CR2","doi-asserted-by":"crossref","unstructured":"Andrade, G., Griebler, D., Santos, R., Danelutto, M., Fernandes, L.G.: Assessing coding metrics for parallel programming of stream processing programs on multi-cores. In: 2021 47th Euromicro Conference on Software Engineering and Advanced Applications (SEAA), pp. 291\u2013295, (2021)","DOI":"10.1109\/SEAA53835.2021.00044"},{"key":"746_CR3","doi-asserted-by":"crossref","unstructured":"Andrade, G., Griebler, D., Santos, R., Fernandes, L.G.: A parallel programming assessment for stream processing applications on multi-core systems. Comput. Stand. Interfaces 84, 103691 (2022)","DOI":"10.1016\/j.csi.2022.103691"},{"key":"746_CR4","doi-asserted-by":"crossref","unstructured":"Andrade, G., Griebler, D., Santos, R., Kessler, C., Ernstsson, A., Fernandes, L.G.: Analyzing programming effort model accuracy of high-level parallel programs for stream processing. In: 2022 48th Euromicro Conference on Software Engineering and Advanced Applications (SEAA), pp. 1\u20134 (2022)","DOI":"10.1109\/SEAA56994.2022.00043"},{"key":"746_CR5","doi-asserted-by":"crossref","unstructured":"Araujo, G., Griebler, D., Rockenbach, D.A., Danelutto, M., Fernandes, L.G.: NAS parallel benchmarks with CUDA and beyond. Softw. Pract. Exp. 1\u201328 (2021)","DOI":"10.1002\/spe.3056"},{"key":"746_CR6","doi-asserted-by":"crossref","unstructured":"Alves de Araujo, G., Griebler, D., Danelutto, M., Fernandes, L.G.: Efficient NAS parallel benchmark kernels with CUDA. In: 2020 28th Euromicro International Conference on Parallel, Distributed and Network-Based Processing (PDP), pp. 9\u201316 (2020)","DOI":"10.1109\/PDP50117.2020.00009"},{"key":"746_CR7","doi-asserted-by":"crossref","unstructured":"Arvanitou, M., Ampatzoglou, A., Nikolaidis, N., Tzintzira, A., Ampatzoglou, A., Chatzigeorgiou, A.: Investigating trade offs between portability, performance and maintainability in exascale systems. In: 46th Euromicro Conference on Software Engineering and Advanced Applications (SEAA), pp. 59\u201363 (2020)","DOI":"10.1109\/SEAA51224.2020.00020"},{"key":"746_CR8","doi-asserted-by":"crossref","unstructured":"Bailey, D.H., Barszcz, E., Barton, J.T., Browning, D.S., Carter, R.L., Dagum, L., Fatoohi, R.A., Frederickson, P.O., Lasinski, T.A., Schreiber, R.S., Simon, H.D. Venkatakrishnan, V., Weeratunga, S.K.: The NAS parallel benchmarks-summary and preliminary results. In: Proceedings of the 1991 ACM\/IEEE Conference on Supercomputing, Supercomputing\u201991, pp. 158\u2013165, New York, NY, USA. Association for Computing Machinery (1991)","DOI":"10.1145\/125826.125925"},{"key":"746_CR9","doi-asserted-by":"crossref","unstructured":"Bienia, C., Kumar, S., Singh, J.P., Li, K.: The PARSEC benchmark suite: Characterization and architectural implications. In: Proceedings of the 17th International Conference on Parallel Architectures and Compilation Techniques, PACT\u201908, pp. 72\u201381, New York, NY, USA. Association for Computing Machinery, (2008)","DOI":"10.1145\/1454115.1454128"},{"key":"746_CR10","doi-asserted-by":"crossref","unstructured":"Che, S., Boyer, M., Meng, J., Tarjan, D., Sheaffer, J.W., Lee, S.-H., Skadron, K.: Rodinia: a benchmark suite for heterogeneous computing. In: 2009 IEEE International Symposium on Workload Characterization (IISWC) pp. 44\u201354 (2009)","DOI":"10.1109\/IISWC.2009.5306797"},{"issue":"3","key":"746_CR11","doi-asserted-by":"publisher","first-page":"389","DOI":"10.1016\/j.parco.2003.12.002","volume":"30","author":"M Cole","year":"2004","unstructured":"Cole, M.: Bringing skeletons out of the closet: a pragmatic manifesto for skeletal parallel programming. Parallel Comput. 30(3), 389\u2013406 (2004)","journal-title":"Parallel Comput."},{"key":"746_CR12","volume-title":"Algorithmic Skeletons: Structured Management of Parallel Computation","author":"MI Cole","year":"1989","unstructured":"Cole, M.I.: Algorithmic Skeletons: Structured Management of Parallel Computation. Pitman and MIT Press, Cambridge (1989)"},{"issue":"3","key":"746_CR13","doi-asserted-by":"publisher","first-page":"506","DOI":"10.1007\/s10766-015-0357-6","volume":"44","author":"U Dastgeer","year":"2016","unstructured":"Dastgeer, U., Kessler, C.: Smart containers and skeleton programming for GPU-based systems. Int. J. Parallel Prog. 44(3), 506\u2013530 (2016)","journal-title":"Int. J. Parallel Prog."},{"key":"746_CR14","doi-asserted-by":"crossref","unstructured":"Dastgeer, U., Li, L., Christoph, K.: Adaptive implementation selection in the SkePU skeleton programming library. In: Revised Selected Papers of the 10th International Symposium on Advanced Parallel Processing Technologies---Volume 8299, APPT 2013, pp. 170\u2013183. Springer, Berlin (2013)","DOI":"10.1007\/978-3-642-45293-2_13"},{"key":"746_CR15","doi-asserted-by":"crossref","unstructured":"De Sensi, D., De Matteis, T., Torquati, M., Mencagli, G., Danelutto, M.: Bringing parallel patterns out of the corner: The P3ARSEC benchmark suite. ACM Trans. Archit. Code Optim. 14(4), 1\u201326 (2017)","DOI":"10.1145\/3132710"},{"key":"746_CR16","doi-asserted-by":"publisher","first-page":"489","DOI":"10.1007\/978-3-319-46079-6_34","volume-title":"High Performance Computing","author":"T Deakin","year":"2016","unstructured":"Deakin, T., James, P., Matt, M., Simon, M.S.: GPU-stream v2.0: benchmarking the achievable memory bandwidth of many-core processors across diverse parallel programming models. In: Taufer, M., Mohr, B., Kunkel, J.M. (eds) High Performance Computing, pp. 489\u2013507. Springer, Cham (2016)"},{"issue":"24","key":"746_CR17","doi-asserted-by":"publisher","first-page":"4175","DOI":"10.1002\/cpe.4175","volume":"29","author":"D del Rio Astorga","year":"2017","unstructured":"del Rio Astorga, D., Dolz, M.F., Fern\u00e1ndez, J., Garc\u00eda, J.D.: A generic parallel pattern interface for stream and data processing. Concurr. Comput. Pract. Exp. 29(24), 4175 (2017)","journal-title":"Concurr. Comput. Pract. Exp."},{"key":"746_CR18","doi-asserted-by":"crossref","unstructured":"Domenico, D.D., Cavalheiro, G.G.H., Lima, J.V.F.: Nas parallel benchmark kernels with python: a performance and programming effort analysis focusing on GPUs. In: 2022 30th Euromicro International Conference on Parallel, Distributed and Network-based Processing (PDP) pp. 26\u201333 (2022)","DOI":"10.1109\/PDP55904.2022.00013"},{"key":"746_CR19","doi-asserted-by":"crossref","unstructured":"Do, Y., Kim, H., Oh, P., Park, D., Lee, J.: SNU-NPB 2019: parallelizing and optimizing NPB in OpenCL and CUDA for modern GPUs. In: 2019 IEEE International Symposium on Workload Characterization (IISWC), pp. 93\u2013105, IEEE (2019)","DOI":"10.1109\/IISWC47752.2019.9041954"},{"key":"746_CR20","doi-asserted-by":"crossref","unstructured":"Do, Y., Kim, H., Oh, P., Park, D., Lee, J.: SNU-NPB 2019: parallelizing and optimizing NPB in OpenCL and CUDA for modern GPUs. In: International Symposium on Workload Characterization (IISWC), pp. 93\u2013105 (2019)","DOI":"10.1109\/IISWC47752.2019.9041954"},{"issue":"2","key":"746_CR21","doi-asserted-by":"publisher","first-page":"129","DOI":"10.1504\/IJHPCN.2012.046370","volume":"7","author":"S Ernsting","year":"2021","unstructured":"Ernsting, S., Kuchen, H.: Algorithmic skeletons for multi-core, multi-GPU systems and clusters. Int. J. High Perform. Comput. Netw. 7(2):129-138 (2012)","journal-title":"Int. J. High Perform. Comput. Netw."},{"key":"746_CR22","doi-asserted-by":"crossref","unstructured":"Ernstsson, A.: Pattern-based Programming Abstractions for Heterogeneous Parallel Computing. PhD thesis, Link\u00f6ping University Electronic Press (2022)","DOI":"10.3384\/9789179291969"},{"key":"746_CR23","doi-asserted-by":"publisher","first-page":"846","DOI":"10.1007\/s10766-021-00704-3","volume":"49","author":"A Ernstsson","year":"2021","unstructured":"Ernstsson, A., Ahlqvist, J., Zouzoula, S., Kessler, C.: SkePU 3: portable high-level programming of heterogeneous systems and HPC clusters. Int. J. Parallel Prog. 49, 846\u2013866 (2021)","journal-title":"Int. J. Parallel Prog."},{"issue":"5","key":"746_CR24","doi-asserted-by":"publisher","first-page":"e5003","DOI":"10.1002\/cpe.5003","volume":"31","author":"A Ernstsson","year":"2019","unstructured":"Ernstsson, A., Kessler, C.: Extending smart containers for data locality-aware skeleton programming. Concurr. Comput. Pract. Exp. 31(5), e5003 (2019)","journal-title":"Concurr. Comput. Pract. Exp."},{"issue":"01","key":"746_CR25","doi-asserted-by":"publisher","first-page":"1740005","DOI":"10.1142\/S0129626417400059","volume":"27","author":"G Griebler","year":"2017","unstructured":"Griebler, D., Danelutto, M., Torquati, M., Fernandes, L.G.: SPar: a DSL for high-level and productive stream parallelism. Parallel Process. Lett. 27(01):1740005 (2017)","journal-title":"Parallel Process. Lett."},{"key":"746_CR26","doi-asserted-by":"crossref","unstructured":"Griebler, D., L\u00f6ff, J., Mencagli, G., Danelutto, M., Fernandes, L.G.: Efficient NAS benchmark kernels with C++ parallel programming. In: 26th Euromicro International Conference on Parallel, Distributed and Network-based Processing (PDP), pp. 733\u2013740 (2018)","DOI":"10.1109\/PDP2018.2018.00120"},{"key":"746_CR27","doi-asserted-by":"publisher","first-page":"743","DOI":"10.1016\/j.future.2021.07.021","volume":"125","author":"J L\u00f6ff","year":"2021","unstructured":"L\u00f6ff, J., Griebler, D., Mencagli, G., Araujo, G., Torquati, M., Danelutto, M., Fernandes, L.G.: The NAS parallel benchmarks for evaluating C++ parallel programming frameworks on shared-memory architectures. Future Gener. Comput. Syst. 125, 743\u2013757 (2021)","journal-title":"Future Gener. Comput. Syst."},{"key":"746_CR28","unstructured":"McCalpin, J.D.: STREAM benchmark (1995)"},{"issue":"4","key":"746_CR29","doi-asserted-by":"publisher","first-page":"792","DOI":"10.1109\/TPDS.2021.3104257","volume":"33","author":"L Papadopoulos","year":"2022","unstructured":"Papadopoulos, L., Soudris, D., Kessler, C., Ernstsson, A., Ahlqvist, J., Vasilas, N., Papadopoulos, A.I., Seferlis, P., Prouveur, C., Haefele, M., Thibault, S., Salamanis, A., Ioakimidis, T., Kehagias, D.: Exa2pro: a framework for high development productivity on heterogeneous computing systems. IEEE Trans. Parallel Distrib. Syst. 33(4), 792\u2013804 (2022)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"746_CR30","doi-asserted-by":"publisher","first-page":"947","DOI":"10.1016\/j.future.2017.08.007","volume":"92","author":"SJ Pennycook","year":"2019","unstructured":"Pennycook, S.J., Sewall, J.D., Lee, V.W.: Implications of a metric for performance portability. Future Gener. Comput. Syst. 92, 947\u2013958 (2019)","journal-title":"Future Gener. Comput. Syst."},{"key":"746_CR31","doi-asserted-by":"crossref","unstructured":"Sewall, J., Pennycook, S.J., Jacobsen, D., Deakin, T., McIntosh-Smith, S.: Interpreting and visualizing performance portability metrics. In: IEEE\/ACM International Workshop on Performance, Portability and Productivity in HPC (P3HPC), pp. 14\u201324 (2020)","DOI":"10.1109\/P3HPC51967.2020.00007"},{"key":"746_CR32","doi-asserted-by":"crossref","unstructured":"Slaughter, E., Wu, W., Fu, Y., Brandenburg, L., Garcia, N., Kautz, W., Marx, E., Morris, K.S., Cao, Q., Bosilca, G., Mirchandaney, S., Leek, W., Treichlerk, S., McCormick, P., Aiken, A.: Task bench: a parameterized benchmark for evaluating parallel runtime performance. In: SC20: International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201315 (2020)","DOI":"10.1109\/SC41405.2020.00066"},{"key":"746_CR33","doi-asserted-by":"crossref","unstructured":"Xu, R., Tian, X., Chandrasekaran, S., Yuan, Y., Chapman, B.: NAS parallel benchmarks for GPGPUs using a directive-based programming model. In: Proceedings of the LCPC 2014, LNCS 8967, pp. 67\u201381. Springer, Berlin (2015)","DOI":"10.1007\/978-3-319-17473-0_5"},{"key":"746_CR34","unstructured":"Yuki, T., Pouchet, L.-N.: Polybench 4.0 (2015)"}],"container-title":["International Journal of Parallel Programming"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-022-00746-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10766-022-00746-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-022-00746-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,27]],"date-time":"2023-01-27T19:40:35Z","timestamp":1674848435000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10766-022-00746-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,6]]},"references-count":34,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2023,2]]}},"alternative-id":["746"],"URL":"https:\/\/doi.org\/10.1007\/s10766-022-00746-1","relation":{},"ISSN":["0885-7458","1573-7640"],"issn-type":[{"value":"0885-7458","type":"print"},{"value":"1573-7640","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,12,6]]},"assertion":[{"value":"10 September 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 November 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 December 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}