{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T21:07:48Z","timestamp":1743109668049,"version":"3.40.3"},"publisher-location":"Cham","reference-count":30,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031506833"},{"type":"electronic","value":"9783031506840"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-50684-0_21","type":"book-chapter","created":{"date-parts":[[2024,4,15]],"date-time":"2024-04-15T10:02:09Z","timestamp":1713175329000},"page":"270-281","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A Performance Analysis of\u00a0Leading Many-Core Technologies for\u00a0Cellular Automata Execution"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7045-2702","authenticated-orcid":false,"given":"Alessio","family":"De Rango","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1701-2203","authenticated-orcid":false,"given":"Donato","family":"D\u2019Ambrosio","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9716-3532","authenticated-orcid":false,"given":"Alfonso","family":"Senatore","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0353-5206","authenticated-orcid":false,"given":"Giuseppe","family":"Mendicino","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1142-3039","authenticated-orcid":false,"given":"Kumudha","family":"Narasimhan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3520-9598","authenticated-orcid":false,"given":"Mehdi","family":"Goli","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5977-4310","authenticated-orcid":false,"given":"Rod","family":"Burns","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,4,16]]},"reference":[{"key":"21_CR1","doi-asserted-by":"publisher","first-page":"258","DOI":"10.1016\/j.jocs.2015.08.009","volume":"11","author":"B Arca","year":"2015","unstructured":"Arca, B., Ghisu, T., Trunfio, G.A.: GPU-accelerated multi-objective optimization of fuel treatments for mitigating wildfire hazard. J. Comput. Sci. 11, 258\u2013268 (2015)","journal-title":"J. Comput. Sci."},{"issue":"1","key":"21_CR2","first-page":"41","volume":"2","author":"M Avolio","year":"2000","unstructured":"Avolio, M., et al.: Simulation of the 1992 Tessina landslide by a cellular automata model and future hazard scenarios. Int. J. Appl. Earth Obs. Geoinf. 2(1), 41\u201350 (2000)","journal-title":"Int. J. Appl. Earth Obs. Geoinf."},{"key":"21_CR3","doi-asserted-by":"crossref","unstructured":"Baratta, I., Richardson, C., Wells, G.: Performance analysis of matrix-free conjugate gradient kernels using SYCL. In: International Workshop on OpenCL. IWOCL\u201922, Association for Computing Machinery, New York, NY, USA (2022)","DOI":"10.1145\/3529538.3529993"},{"key":"21_CR4","doi-asserted-by":"publisher","first-page":"120","DOI":"10.1016\/j.jpdc.2022.03.017","volume":"165","author":"G Casta\u00f1o","year":"2022","unstructured":"Casta\u00f1o, G., Faqir-Rhazoui, Y., Garc\u00eda, C., Prieto-Mat\u00edas, M.: Evaluation of Intel\u2019s DPC++ compatibility tool in heterogeneous computing. J. Parall. Distrib. Comput. 165, 120\u2013129 (2022)","journal-title":"J. Parall. Distrib. Comput."},{"key":"21_CR5","doi-asserted-by":"publisher","first-page":"295","DOI":"10.1016\/j.cpc.2015.01.026","volume":"192","author":"J Cercos-Pita","year":"2015","unstructured":"Cercos-Pita, J.: AQUAgpusph, a new free 3D SPH solver accelerated with OpenCL. Comput. Phys. Commun. 192, 295\u2013312 (2015)","journal-title":"Comput. Phys. Commun."},{"key":"21_CR6","doi-asserted-by":"publisher","first-page":"53","DOI":"10.1016\/j.jpdc.2018.07.005","volume":"121","author":"D D\u2019Ambrosio","year":"2018","unstructured":"D\u2019Ambrosio, D., et al.: The open computing abstraction layer for parallel complex systems modeling on many-core systems. J. Parall. Distrib. Comput. 121, 53\u201370 (2018)","journal-title":"J. Parall. Distrib. Comput."},{"issue":"2","key":"21_CR7","doi-asserted-by":"publisher","first-page":"630","DOI":"10.1007\/s11227-013-0949-0","volume":"65","author":"D D\u2019Ambrosio","year":"2013","unstructured":"D\u2019Ambrosio, D., Filippone, G., Marocco, D., Rongo, R., Spataro, W.: Efficient application of GPGPU for lava flow hazard mapping. J. Supercomput. 65(2), 630\u2013644 (2013)","journal-title":"J. Supercomput."},{"issue":"3","key":"21_CR8","doi-asserted-by":"publisher","first-page":"30","DOI":"10.4018\/jghpc.2012070102","volume":"4","author":"D D\u2019Ambrosio","year":"2012","unstructured":"D\u2019Ambrosio, D., Filippone, G., Rongo, R., Spataro, W., Trunfio, G.: Cellular automata and GPGPU: an application to lava flow modeling. Int. J. Grid and High Perform. Comput. 4(3), 30\u201347 (2012)","journal-title":"Int. J. Grid and High Perform. Comput."},{"key":"21_CR9","unstructured":"D\u2019Ambrosio, D., Terremoto, G., De Rango, A., Furnari, L., Senatore, A., Mendicino, G.: First SYCL implementation of the three-dimensional subsurface XCA-Flow cellular automaton and performance comparison against CUDA. In: Proceedings of the International Conference on Applied Computing 2022 and WWW\/Internet 2022, Lisbon, Portugal, 8\u20139 November, 2022, pp. 47\u201354 (2022)"},{"key":"21_CR10","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1007\/978-3-030-39081-5_22","volume-title":"Numerical Computations: Theory and Algorithms","author":"D D\u2019Ambrosio","year":"2020","unstructured":"D\u2019Ambrosio, D., et al.: A general computational formalism for networks of structured grids. In: Sergeyev, Y.D., Kvasov, D.E. (eds.) Numerical Computations: Theory and Algorithms, pp. 243\u2013255. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-39081-5_22"},{"key":"21_CR11","unstructured":"De Rango, A., D\u2019Ambrosio, D.: Github repository of developed software. https:\/\/github.com\/alessioderango\/CUDA_OpenCL_SYCL_Perfomance_Assessment"},{"key":"21_CR12","unstructured":"Deakin, T., McIntosh-Smith, S.: Evaluating the performance of HPC-style SYCL applications. In: IWOCL \u201920. Association for Computing Machinery (ACM), United States (2020), international Workshop on OpenCL, IWOCL; Conference date: 27\u201304-2020 Through 29\u201304-2020"},{"key":"21_CR13","doi-asserted-by":"crossref","unstructured":"Giordano, A., De Rango, A., D\u2019Ambrosio, D., Rongo, R., Spataro, W.: Strategies for parallel execution of cellular automata in distributed memory architectures. In: 2019 27th Euromicro International Conference on Parallel, Distributed and Network-Based Processing (PDP), pp. 406\u2013413 (2019)","DOI":"10.1109\/EMPDP.2019.8671639"},{"issue":"2","key":"21_CR14","doi-asserted-by":"publisher","first-page":"470","DOI":"10.1109\/TPDS.2020.3025102","volume":"32","author":"A Giordano","year":"2021","unstructured":"Giordano, A., De Rango, A., Rongo, R., D\u2019Ambrosio, D., Spataro, W.: Dynamic load balancing in parallel execution of cellular automata. IEEE Trans. Parallel Distrib. Syst. 32(2), 470\u2013484 (2021)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"21_CR15","doi-asserted-by":"crossref","unstructured":"Goli, M., et al.: Towards cross-platform performance portability of DNN models using SYCL. In: Proceedings of P3HPC 2020: International Workshop on Performance, Portability, and Productivity in HPC, Held in conjunction with SC 2020: The International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 25\u201335 (2020)","DOI":"10.1109\/P3HPC51967.2020.00008"},{"key":"21_CR16","doi-asserted-by":"publisher","first-page":"259","DOI":"10.1016\/S0167-739X(99)00051-5","volume":"16","author":"SD Gregorio","year":"1999","unstructured":"Gregorio, S.D., Serra, R.: An empirical method for modelling and simulating some complex macroscopic phenomena by cellular automata. Futur. Gener. Comput. Syst. 16, 259\u2013271 (1999)","journal-title":"Futur. Gener. Comput. Syst."},{"key":"21_CR17","doi-asserted-by":"crossref","unstructured":"Kirk, D.B., Mei, W., Hwu, W.: Chapter 7 - parallel patterns: convolution: an introduction to stencil computation. In: Kirk, D.B., Mei W. Hwu, W. (eds.) Programming Massively Parallel Processors (Third Edition), pp. 149\u2013174. Morgan Kaufmann, third edition (2017)","DOI":"10.1016\/B978-0-12-811986-0.00007-8"},{"key":"21_CR18","doi-asserted-by":"crossref","unstructured":"Konstantinidis, E., Cotronis, Y.: A Quantitative performance evaluation of fast on-chip memories of GPUs. In: 2016 24th Euromicro International Conference on Parallel, Distributed, and Network-Based Processing (PDP), pp. 448\u2013455 (2016)","DOI":"10.1109\/PDP.2016.56"},{"key":"21_CR19","doi-asserted-by":"crossref","unstructured":"Macri, M., De Rango, A., Spataro, D., D\u2019Ambrosio, D., Spataro, W.: Efficient lava flows simulations with OpenCL: a preliminary application for civil defence purposes. In: Proceedings of the 10th International Conference on P2P, Parallel, Grid, Cloud and Internet Computing, 3PGCIC 2015, pp. 328\u2013335 (2015)","DOI":"10.1109\/3PGCIC.2015.107"},{"key":"21_CR20","volume-title":"Theory of Self-Reproducing Automata","author":"J von Neumann","year":"1966","unstructured":"von Neumann, J.: Theory of Self-Reproducing Automata. University of Illinois Press, Champaign, IL, USA (1966)"},{"issue":"2","key":"21_CR21","doi-asserted-by":"publisher","first-page":"40","DOI":"10.1145\/1365490.1365500","volume":"6","author":"J Nickolls","year":"2008","unstructured":"Nickolls, J., Buck, I., Garland, M., Skadron, K.: Scalable parallel programming with CUDA: Is CUDA the parallel programming model that application developers have been waiting for? Queue 6(2), 40\u201353 (2008)","journal-title":"Queue"},{"issue":"1","key":"21_CR22","doi-asserted-by":"publisher","first-page":"80","DOI":"10.1111\/j.1467-8659.2007.01012.x","volume":"26","author":"J Owens","year":"2007","unstructured":"Owens, J., et al.: A survey of general-purpose computation on graphics hardware. Comput. Graph. Forum 26(1), 80\u2013113 (2007)","journal-title":"Comput. Graph. Forum"},{"issue":"1","key":"21_CR23","doi-asserted-by":"publisher","first-page":"174","DOI":"10.1109\/TPDS.2018.2855182","volume":"30","author":"B Peccerillo","year":"2019","unstructured":"Peccerillo, B., Bartolini, S.: PHAST - a portable high-level modern C++ programming library for GPUs and multi-cores. IEEE Trans. Parall. Distrib. Syst. 30(1), 174\u2013189 (2019)","journal-title":"IEEE Trans. Parall. Distrib. Syst."},{"key":"21_CR24","doi-asserted-by":"publisher","first-page":"115","DOI":"10.1016\/j.jocs.2018.09.012","volume":"32","author":"AD Rango","year":"2019","unstructured":"Rango, A.D., Spataro, D., Spataro, W., D\u2019Ambrosio, D.: A first multi-GPU\/multi-node implementation of the open computing abstraction layer. J. Comput. Sci. 32, 115\u2013124 (2019)","journal-title":"J. Comput. Sci."},{"key":"21_CR25","doi-asserted-by":"crossref","unstructured":"Reguly, I., Owenson, A., Powell, A., Jarvis, S., Mudalige, G.: Under the Hood of SYCL - an initial performance analysis with an unstructured-mesh CFD application. In: Proceedings - International Supercomputing Conference (ISC21) (2021)","DOI":"10.1007\/978-3-030-78713-4_21"},{"key":"21_CR26","doi-asserted-by":"crossref","unstructured":"Reyes, R., Brown, G., Burns, R., Wong, M.: SYCL 2020: more than meets the eye. In: Proceedings of the International Workshop on OpenCL. IWOCL \u201920, Association for Computing Machinery, New York, NY, USA (2020)","DOI":"10.1145\/3388333.3388649"},{"issue":"3","key":"21_CR27","doi-asserted-by":"publisher","first-page":"66","DOI":"10.1109\/MCSE.2010.69","volume":"12","author":"J Stone","year":"2010","unstructured":"Stone, J., Gohara, D., Shi, G.: OpenCL: a parallel programming standard for heterogeneous computing systems. Comput. Sci. Eng. 12(3), 66\u201372 (2010)","journal-title":"Comput. Sci. Eng."},{"issue":"4","key":"21_CR28","doi-asserted-by":"publisher","first-page":"65","DOI":"10.1145\/1498765.1498785","volume":"52","author":"S Williams","year":"2009","unstructured":"Williams, S., Waterman, A., Patterson, D.: Roofline: an insightful visual performance model for multicore architectures. Commun. ACM 52(4), 65\u201376 (2009)","journal-title":"Commun. ACM"},{"key":"21_CR29","unstructured":"Yang, C.: Hierarchical roofline analysis: how to collect data using performance tools on intel CPUs and NVIDIA GPUs (2020). https:\/\/arxiv.org\/abs\/2009.02449"},{"issue":"20","key":"21_CR30","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.5547","volume":"32","author":"C Yang","year":"2020","unstructured":"Yang, C., Kurth, T., Williams, S.: Hierarchical roofline analysis for GPUs: accelerating performance optimization for the NERSC-9 Perlmutter system. Concurr. Comput.: Pract. Exper. 32(20), e5547 (2020)","journal-title":"Concurr. Comput.: Pract. Exper."}],"container-title":["Lecture Notes in Computer Science","Euro-Par 2023: Parallel Processing Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-50684-0_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,7]],"date-time":"2024-05-07T17:04:12Z","timestamp":1715101452000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-50684-0_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031506833","9783031506840"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-50684-0_21","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"16 April 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"Euro-Par","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Parallel Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Limassol","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Cyprus","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 August 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 September 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"europar2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2023.euro-par.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"easychair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"164","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"49","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"30% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.98","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}