{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:39:16Z","timestamp":1740123556607,"version":"3.37.3"},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"13","license":[{"start":{"date-parts":[[2024,5,23]],"date-time":"2024-05-23T00:00:00Z","timestamp":1716422400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,5,23]],"date-time":"2024-05-23T00:00:00Z","timestamp":1716422400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100003593","name":"Conselho Nacional de Desenvolvimento Cient\u00edfico e Tecnol\u00f3gico","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100003593","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002322","name":"Coordena\u00e7\u00e3o de Aperfei\u00e7oamento de Pessoal de N\u00edvel Superior","doi-asserted-by":"publisher","award":["001"],"award-info":[{"award-number":["001"]}],"id":[{"id":"10.13039\/501100002322","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1007\/s11227-024-06079-9","type":"journal-article","created":{"date-parts":[[2024,5,23]],"date-time":"2024-05-23T17:01:48Z","timestamp":1716483708000},"page":"18838-18865","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Allok: a machine learning approach for efficient graph execution on CPU\u2013GPU clusters"],"prefix":"10.1007","volume":"80","author":[{"given":"Marcelo Koji","family":"Moori","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hiago Mayk G. de A.","family":"Rocha","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arthur F.","family":"Lorenzon","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Antonio Carlos S.","family":"Beck","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,5,23]]},"reference":[{"issue":"1","key":"6079_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3434393","volume":"8","author":"L Dhulipala","year":"2021","unstructured":"Dhulipala L, Blelloch GE, Shun J (2021) Theoretically efficient parallel graph algorithms can be fast and scalable. ACM Trans Parallel Comput (TOPC) 8(1):1\u201370","journal-title":"ACM Trans Parallel Comput (TOPC)"},{"issue":"1","key":"6079_CR2","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3466795","volume":"48","author":"C Yang","year":"2022","unstructured":"Yang C, Bulu\u00e7 A, Owens JD (2022) GraphBLAST: a high-performance linear algebra-based graph framework on the GPU. ACM Trans Math Softw (TOMS) 48(1):1\u201351","journal-title":"ACM Trans Math Softw (TOMS)"},{"issue":"4","key":"6079_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3322125","volume":"45","author":"TA Davis","year":"2019","unstructured":"Davis TA (2019) Algorithm 1000: SuiteSparse: GraphBLAS: graph algorithms in the language of sparse linear algebra. ACM Trans Math Softw (TOMS) 45(4):1\u201325","journal-title":"ACM Trans Math Softw (TOMS)"},{"doi-asserted-by":"crossref","unstructured":"Khorasani F, Vora K, Gupta R, Bhuyan LN (2014) CuSha: vertex-centric graph processing on GPUs. In: Proceedings of the 23rd International Symposium on High-performance Parallel and Distributed Computing, pp 239\u2013252","key":"6079_CR4","DOI":"10.1145\/2600212.2600227"},{"doi-asserted-by":"crossref","unstructured":"Shun J, Blelloch GE (2013) Ligra: a lightweight graph processing framework for shared memory. In: Proceedings of the 18th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp 135\u2013146","key":"6079_CR5","DOI":"10.1145\/2442516.2442530"},{"doi-asserted-by":"crossref","unstructured":"Sun J, Vandierendonck H, Nikolopoulos DS (2017) Graphgrind: addressing load imbalance of graph partitioning. In: Proceedings of the International Conference on Supercomputing, pp 1\u201310","key":"6079_CR6","DOI":"10.1145\/3079079.3079097"},{"key":"6079_CR7","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-28719-1","volume-title":"Parallel computing hits the power wall: principles, challenges, and a survey of solutions","author":"AF Lorenzon","year":"2019","unstructured":"Lorenzon AF, Filho ACSB (2019) Parallel computing hits the power wall: principles, challenges, and a survey of solutions. Springer, Berlin"},{"doi-asserted-by":"crossref","unstructured":"de\u00a0Rocha HMGA, Schwarzrock J, Lorenzon AF, Beck ACS (2021) Boosting graph analytics by tuning threads and data affinity on numa systems. In: 2021 29th Euromicro International Conference on Parallel, Distributed and Network-Based Processing (PDP). IEEE, pp 161\u2013168","key":"6079_CR8","DOI":"10.1109\/PDP52278.2021.00033"},{"issue":"6","key":"6079_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3128571","volume":"50","author":"X Shi","year":"2018","unstructured":"Shi X, Zheng Z, Zhou Y, Jin H, He L, Liu B, Hua Q-S (2018) Graph processing on GPUs: a survey. ACM Comput Surv (CSUR) 50(6):1\u201335","journal-title":"ACM Comput Surv (CSUR)"},{"doi-asserted-by":"crossref","unstructured":"Moori MK, de\u00a0Rocha HMGA, Silva MA, Schwarzrock J, Lorenzon AF, Beck ACS (2023) Automatic CPU\u2013GPU allocation for graph execution. In: 2023 31st Euromicro International Conference on Parallel, Distributed and Network-Based Processing (PDP). IEEE, pp 27\u201334","key":"6079_CR10","DOI":"10.1109\/PDP59025.2023.00013"},{"doi-asserted-by":"crossref","unstructured":"de\u00a0Rocha HMGA, Schwarzrock J, Lorenzon AF, Beck ACS (2022) Using machine learning to optimize graph execution on numa machines. In: Proceedings of the 59th ACM\/IEEE Design Automation Conference, pp 1027\u20131032","key":"6079_CR11","DOI":"10.1145\/3489517.3530581"},{"issue":"18","key":"6079_CR12","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.7419","volume":"35","author":"MK Moori","year":"2023","unstructured":"Moori MK, de Rocha HMGA, Schwarzrock J, Lorenzon AF, Beck AS (2023) Improving the efficiency of graph algorithm executions on high-performance computing. Concurr Comput Pract Exp 35(18):e7419","journal-title":"Concurr Comput Pract Exp"},{"doi-asserted-by":"crossref","unstructured":"Rocha HMGDA, Querol VB, Lorenzon AF, Beck ACS (2023) Optimizing single-source graph execution on NUMA machines. In: XIII Brazilian Symposium on Computing Systems Engineering (SBESC). IEEE, pp 1\u20136","key":"6079_CR13","DOI":"10.1109\/SBESC60926.2023.10324068"},{"unstructured":"Leskovec J, Krevl A (2014) SNAP datasets: stanford large network dataset collection. http:\/\/snap.stanford.edu\/data","key":"6079_CR14"},{"key":"6079_CR15","first-page":"11","volume":"1695","author":"G Csardi","year":"2005","unstructured":"Csardi G, Nepusz T (2005) The igraph software package for complex network research. InterJ Complex Syst 1695:11","journal-title":"InterJ Complex Syst"},{"issue":"4","key":"6079_CR16","doi-asserted-by":"publisher","first-page":"508","DOI":"10.1017\/nws.2016.20","volume":"4","author":"CL Staudt","year":"2016","unstructured":"Staudt CL, Sazonovs A, Meyerhenke H (2016) Networkit: a tool suite for large-scale complex network analysis. Netw Sci 4(4):508\u2013530","journal-title":"Netw Sci"},{"doi-asserted-by":"crossref","unstructured":"Nguyen D, Lenharth A, Pingali K (2013) A lightweight infrastructure for graph analytics. In: ACM SOSP, pp 456\u2013471","key":"6079_CR17","DOI":"10.1145\/2517349.2522739"},{"issue":"2","key":"6079_CR18","doi-asserted-by":"publisher","first-page":"163","DOI":"10.1080\/0022250X.2001.9990249","volume":"25","author":"U Brandes","year":"2001","unstructured":"Brandes U (2001) A faster algorithm for betweenness centrality. J Math Sociol 25(2):163\u2013177","journal-title":"J Math Sociol"},{"doi-asserted-by":"crossref","unstructured":"Madduri K, Ediger D, Jiang K, Bader DA, Chavarria-Miranda D (2009) A faster parallel algorithm and efficient multithreaded implementations for evaluating betweenness centrality on massive datasets. In: 2009 IEEE International Symposium on Parallel and Distributed Processing, pp 1\u20138","key":"6079_CR19","DOI":"10.1109\/IPDPS.2009.5161100"},{"doi-asserted-by":"crossref","unstructured":"Beamer S, Asanovic K, Patterson D (2012) Direction-optimizing breadth-first search. In: SC \u201912: Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis, pp 1\u201310","key":"6079_CR20","DOI":"10.1109\/SC.2012.50"},{"issue":"1","key":"6079_CR21","doi-asserted-by":"publisher","first-page":"114","DOI":"10.1016\/S0196-6774(03)00076-2","volume":"49","author":"U Meyer","year":"2003","unstructured":"Meyer U, Sanders P (2003) delta-stepping: a parallelizable shortest path algorithm. J Algorithms 49(1):114\u2013152","journal-title":"J Algorithms"},{"unstructured":"Beamer S, Asanovi\u0107 K, Patterson D (2015) The gap benchmark suite. arXiv preprint arXiv:1508.03619","key":"6079_CR22"},{"doi-asserted-by":"crossref","unstructured":"Wang Y, Davidson A, Pan Y, Wu Y, Riffel A, Owens JD (2016) Gunrock: a high-performance graph processing library on the GPU. In: Proceedings of the 21st ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp 1\u201312,","key":"6079_CR23","DOI":"10.1145\/2851141.2851145"},{"doi-asserted-by":"crossref","unstructured":"Roy A, Mihailovic I, Zwaenepoel W (2013) X-stream: Edge-centric graph processing using streaming partitions. In: Proceedings of the Twenty-Fourth ACM Symposium on Operating Systems Principles, pp 472\u2013488","key":"6079_CR24","DOI":"10.1145\/2517349.2522740"},{"doi-asserted-by":"crossref","unstructured":"Zhang K, Chen R, Chen H (2015) Numa-aware graph-structured analytics. In: ACM PPoPP, pp 183\u2013193","key":"6079_CR25","DOI":"10.1145\/2858788.2688507"},{"doi-asserted-by":"crossref","unstructured":"Pingali K, Nguyen D, Kulkarni M, Burtscher M, Hassaan MA, Kaleem R, Lee T-H, Lenharth A, Manevich R, M\u00e9ndez-Lojo M et\u00a0al. (2011) The tao of parallelism in algorithms. In: ACM SIGPLAN, pp 12\u201325","key":"6079_CR26","DOI":"10.1145\/1993316.1993501"},{"key":"6079_CR27","doi-asserted-by":"publisher","DOI":"10.1201\/b16251","volume-title":"Parallel science and engineering applications: the Charm++ approach","author":"LV Kale","year":"2016","unstructured":"Kale LV, Bhatele A (2016) Parallel science and engineering applications: the Charm++ approach. CRC Press, Boca Raton"},{"doi-asserted-by":"crossref","unstructured":"Harish P, Narayanan PJ (2007) Accelerating large graph algorithms on the GPU using CUDA. In: HiPC. Springer, pp 197\u2013208","key":"6079_CR28","DOI":"10.1007\/978-3-540-77220-0_21"},{"issue":"8","key":"6079_CR29","doi-asserted-by":"publisher","first-page":"267","DOI":"10.1145\/2038037.1941590","volume":"46","author":"S Hong","year":"2011","unstructured":"Hong S, Kim SK, Oguntebi T, Olukotun K (2011) Accelerating CUDA graph algorithms at maximum warp. Acm Sigplan Notices 46(8):267\u2013276","journal-title":"Acm Sigplan Notices"},{"doi-asserted-by":"crossref","unstructured":"Luo L, Wong M, Hwu W-M (2010) An effective GPU implementation of breadth-first search. In: DAC. IEEE, pp 52\u201355","key":"6079_CR30","DOI":"10.1145\/1837274.1837289"},{"doi-asserted-by":"crossref","unstructured":"Davidson A, Baxter S, Garland M, Owens JD (2014) Work-efficient parallel GPU methods for single-source shortest paths. In: IPDPS. IEEE, pp 349\u2013359","key":"6079_CR31","DOI":"10.1109\/IPDPS.2014.45"},{"doi-asserted-by":"crossref","unstructured":"Wang W, Davidson JW, Soffa ML (2016) Predicting the memory bandwidth and optimal core allocations for multi-threaded applications on large-scale numa machines. In: 2016 IEEE International Symposium on High Performance Computer Architecture. IEEE, pp 419\u2013431","key":"6079_CR32","DOI":"10.1109\/HPCA.2016.7446083"},{"issue":"5","key":"6079_CR33","doi-asserted-by":"publisher","first-page":"1007","DOI":"10.1109\/TPDS.2018.2872992","volume":"30","author":"AF Lorenzon","year":"2018","unstructured":"Lorenzon AF, De Oliveira CC, Souza JD, Beck ACS (2018) Aurora: seamless optimization of openmp applications. IEEE Trans Parallel Distrib Syst 30(5):1007\u20131021","journal-title":"IEEE Trans Parallel Distrib Syst"},{"issue":"7","key":"6079_CR34","doi-asserted-by":"publisher","first-page":"1713","DOI":"10.1109\/TPDS.2020.3046537","volume":"32","author":"J Schwarzrock","year":"2020","unstructured":"Schwarzrock J, De Oliveira CC, Ritt M, Lorenzon AF, Filho ACSB (2020) A runtime and non-intrusive approach to optimize EDP by tuning threads and CPU frequency for openmp applications. IEEE Trans Parallel Distrib Syst 32(7):1713\u20131724","journal-title":"IEEE Trans Parallel Distrib Syst"},{"doi-asserted-by":"crossref","unstructured":"Rossi R, Ahmed N (2015) The network data repository with interactive graph analytics and visualization. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol 29","key":"6079_CR35","DOI":"10.1609\/aaai.v29i1.9277"},{"issue":"4","key":"6079_CR36","doi-asserted-by":"publisher","first-page":"420","DOI":"10.1145\/3186728.3164139","volume":"11","author":"S Sahu","year":"2017","unstructured":"Sahu S, Mhedhbi A, Salihoglu S, Lin J, \u00d6zsu MT (2017) The ubiquity of large graphs and surprising challenges of graph processing. Proc VLDB Endow 11(4):420\u2013431","journal-title":"Proc VLDB Endow"},{"unstructured":"Zhang A, Lipton ZC, Li M, Smola AJ (2021) Dive into deep learning. arXiv preprint arXiv:2106.11342","key":"6079_CR37"},{"key":"6079_CR38","doi-asserted-by":"publisher","first-page":"421","DOI":"10.1007\/978-3-642-35289-8_25","volume-title":"Neural networks: tricks of the trade","author":"L Bottou","year":"2012","unstructured":"Bottou L (2012) Stochastic gradient descent tricks. In: Montavon G, Orr GB, M\u00fcller KR (eds) Neural networks: tricks of the trade, 2nd edn. Springer, Berlin, pp 421\u2013436","edition":"2"},{"doi-asserted-by":"crossref","unstructured":"Zhang Z (2018) Improved adam optimizer for deep neural networks. In: 2018 IEEE\/ACM 26th International Symposium on Quality of Service (IWQoS). IEEE, pp 1\u20132","key":"6079_CR39","DOI":"10.1109\/IWQoS.2018.8624183"},{"doi-asserted-by":"crossref","unstructured":"Lorenzon AF, Cera MC, Beck ACS (2015) On the influence of static power consumption in multicore embedded systems. In: ISCAS. IEEE, pp 1374\u20131377","key":"6079_CR40","DOI":"10.1109\/ISCAS.2015.7168898"},{"issue":"1","key":"6079_CR41","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3434402","volume":"18","author":"PS Labini","year":"2021","unstructured":"Labini PS, Cianfriglia M, Perri D, Gervasi O, Fursin G, Lokhmotov A, Nugteren C, Carpentieri B, Zollo F, Vella F (2021) On the anatomy of predictive models for accelerating GPU convolution kernels and beyond. ACM Trans Archit Code Optim (TACO) 18(1):1\u201324","journal-title":"ACM Trans Archit Code Optim (TACO)"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06079-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-024-06079-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06079-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,25]],"date-time":"2024-07-25T10:15:06Z","timestamp":1721902506000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-024-06079-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,23]]},"references-count":41,"journal-issue":{"issue":"13","published-print":{"date-parts":[[2024,9]]}},"alternative-id":["6079"],"URL":"https:\/\/doi.org\/10.1007\/s11227-024-06079-9","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"type":"print","value":"0920-8542"},{"type":"electronic","value":"1573-0484"}],"subject":[],"published":{"date-parts":[[2024,5,23]]},"assertion":[{"value":"18 March 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 May 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}