{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T06:27:40Z","timestamp":1742970460625,"version":"3.40.3"},"publisher-location":"Cham","reference-count":28,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030024642"},{"type":"electronic","value":"9783030024659"}],"license":[{"start":{"date-parts":[[2018,1,1]],"date-time":"2018-01-01T00:00:00Z","timestamp":1514764800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018]]},"DOI":"10.1007\/978-3-030-02465-9_37","type":"book-chapter","created":{"date-parts":[[2019,1,24]],"date-time":"2019-01-24T20:06:48Z","timestamp":1548360408000},"page":"527-539","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["How Pre-multicore Methods and Algorithms Perform in Multicore Era"],"prefix":"10.1007","author":[{"given":"Alexey","family":"Lastovetsky","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Muhammad","family":"Fahad","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hamidreza","family":"Khaleghzadeh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Semyon","family":"Khokhriakov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ravi","family":"Reddy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arsalan","family":"Shahid","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lukasz","family":"Szustak","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Roman","family":"Wyrzykowski","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,1,25]]},"reference":[{"key":"37_CR1","doi-asserted-by":"publisher","unstructured":"Fatica, M.: Accelerating Linpack with CUDA on heterogenous clusters. In: GPGPU-2, pp. 46\u201351. ACM (2009). \n                      https:\/\/doi.org\/10.1145\/1513895.1513901","DOI":"10.1145\/1513895.1513901"},{"key":"37_CR2","doi-asserted-by":"crossref","unstructured":"Yang, C., Wang, F., Du, Y., et al.: Adaptive optimization for petascale heterogeneous CPU\/GPU computing. In: Cluster 2010, pp. 19\u201328 (2010)","DOI":"10.1109\/CLUSTER.2010.12"},{"key":"37_CR3","unstructured":"Ogata, Y., Endo, T., Maruyama, N., Matsuoka, S.: An efficient, model-based CPU-GPU heterogeneous FFT library. In: IPDPS 2008, pp. 1\u201310 (2008)"},{"issue":"1","key":"37_CR4","doi-asserted-by":"publisher","first-page":"76","DOI":"10.1177\/1094342006074864","volume":"21","author":"A Lastovetsky","year":"2007","unstructured":"Lastovetsky, A., Reddy, R.: Data partitioning with a functional performance model of heterogeneous processors. Int. J. High Perform. Comput. Appl. 21(1), 76\u201390 (2007)","journal-title":"Int. J. High Perform. Comput. Appl."},{"key":"37_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"445","DOI":"10.1007\/978-3-319-21909-7_43","volume-title":"Parallel Computing Technologies","author":"K Rojek","year":"2015","unstructured":"Rojek, K., Wyrzykowski, R.: Parallelization of 3D MPDATA algorithm using many graphics processors. In: Malyshkin, V. (ed.) PaCT 2015. LNCS, vol. 9251, pp. 445\u2013457. Springer, Cham (2015). \n                      https:\/\/doi.org\/10.1007\/978-3-319-21909-7_43"},{"issue":"9","key":"37_CR6","doi-asserted-by":"publisher","first-page":"2506","DOI":"10.1109\/TC.2014.2375202","volume":"64","author":"Z Zhong","year":"2015","unstructured":"Zhong, Z., Rychkov, V., Lastovetsky, A.: Data partitioning on multicore and multi-GPU platforms using functional performance models. IEEE Trans. Comput. 64(9), 2506\u20132518 (2015)","journal-title":"IEEE Trans. Comput."},{"key":"37_CR7","doi-asserted-by":"publisher","first-page":"287","DOI":"10.1145\/1353536.1346318","volume":"43","author":"MD Linderman","year":"2008","unstructured":"Linderman, M.D., Collins, J.D., Wang, H., et al.: Merge: a programming model for heterogeneous multi-core systems. SIGPLAN Not. 43, 287\u2013296 (2008)","journal-title":"SIGPLAN Not."},{"key":"37_CR8","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1007\/978-3-642-14122-5_9","volume-title":"Euro-Par 2009 \u2013 Parallel Processing Workshops","author":"C Augonnet","year":"2010","unstructured":"Augonnet, C., Thibault, S., Namyst, R.: Automatic calibration of performance models on heterogeneous multicore architectures. In: Lin, H.-X., et al. (eds.) Euro-Par 2009. LNCS, vol. 6043, pp. 56\u201365. Springer, Heidelberg (2010). \n                      https:\/\/doi.org\/10.1007\/978-3-642-14122-5_9"},{"key":"37_CR9","doi-asserted-by":"publisher","first-page":"121","DOI":"10.1145\/1594835.1504196","volume":"44","author":"G Quintana-Ort\u00ed","year":"2009","unstructured":"Quintana-Ort\u00ed, G., Igual, F.D., Quintana-Ort\u00ed, E.S., van de Geijn, R.A.: Solving dense linear systems on platforms with multiple hardware accelerators. SIGPLAN Not. 44, 121\u2013130 (2009)","journal-title":"SIGPLAN Not."},{"issue":"3","key":"37_CR10","doi-asserted-by":"publisher","first-page":"787","DOI":"10.1109\/TPDS.2016.2599527","volume":"28","author":"A Lastovetsky","year":"2017","unstructured":"Lastovetsky, A., Szustak, L., Wyrzykowski, R.: Model-based optimization of EULAG kernel on Intel Xeon Phi through load imbalancing. IEEE Trans. Parallel Distrib. Syst. 28(3), 787\u2013797 (2017)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"37_CR11","doi-asserted-by":"crossref","unstructured":"Luk, C.K., Hong, S., Kim, H.: Qilin: exploiting parallelism on heterogeneous multiprocessors with adaptive mapping. In: MICRO-42, pp. 45\u201355 (2009)","DOI":"10.1145\/1669112.1669121"},{"key":"37_CR12","doi-asserted-by":"publisher","first-page":"356","DOI":"10.1093\/comjnl\/40.6.356","volume":"40","author":"M Cierniak","year":"1997","unstructured":"Cierniak, M., Zaki, M., Li, W.: Compile-time scheduling algorithms for heterogeneous network of workstations. Comput. J. 40, 356\u2013372 (1997)","journal-title":"Comput. J."},{"issue":"4","key":"37_CR13","doi-asserted-by":"publisher","first-page":"520","DOI":"10.1006\/jpdc.2000.1686","volume":"61","author":"A Kalinov","year":"2001","unstructured":"Kalinov, A., Lastovetsky, A.: Heterogeneous distribution of computations solving linear algebra problems on networks of heterogeneous computers. J. Parallel Distrib. Comput. 61(4), 520\u2013535 (2001)","journal-title":"J. Parallel Distrib. Comput."},{"issue":"2","key":"37_CR14","doi-asserted-by":"publisher","first-page":"151","DOI":"10.1007\/s11227-009-0350-1","volume":"58","author":"J Mart\u00ednez","year":"2011","unstructured":"Mart\u00ednez, J., Garz\u00f3n, E., Plaza, A., Garc\u00eda, I.: Automatic tuning of iterative computation on heterogeneous multiprocessors with ADITHE. J. Supercomput. 58(2), 151\u2013159 (2011)","journal-title":"J. Supercomput."},{"key":"37_CR15","series-title":"IFIP \u2014 The International Federation for Information Processing","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1007\/0-387-24049-7_3","volume-title":"High Performance Computational Science and Engineering","author":"A Lastovetsky","year":"2005","unstructured":"Lastovetsky, A., Twamley, J.: Towards a realistic performance model for networks of heterogeneous computers. In: Ng, M.K., Doncescu, A., Yang, L.T., Leng, T. (eds.) High Performance Computational Science and Engineering. ITIFIP, vol. 172, pp. 39\u201357. Springer, Boston, MA (2005). \n                      https:\/\/doi.org\/10.1007\/0-387-24049-7_3"},{"key":"37_CR16","unstructured":"Lastovetsky, A., Reddy, R.: Data partitioning with a realistic performance model of networks of heterogeneous computers. In: Proceedings of the 18th International Parallel and Distributed Processing Symposium (IPDPS 2004). IEEE Computer Society, Santa Fe (2004)"},{"key":"37_CR17","unstructured":"Ilic, A., Pratas, F., Trancoso, P., Sousa, L.: High-performance computing on heterogeneous systems: database queries on CPU and GPU. In: High Performance Scientific Computing with Special Emphasis on Current Capabilities and Future Perspectives. IOS Press, Amsterdam (2011)"},{"key":"37_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"693","DOI":"10.1007\/978-3-642-55224-3_65","volume-title":"Parallel Processing and Applied Mathematics","author":"J Cola\u00e7o","year":"2014","unstructured":"Cola\u00e7o, J., Matoga, A., Ilic, A., Roma, N., Tom\u00e1s, P., Chaves, R.: Transparent application acceleration by intelligent scheduling of shared library calls on heterogeneous systems. In: Wyrzykowski, R., Dongarra, J., Karczewski, K., Wa\u015bniewski, J. (eds.) PPAM 2013, part I. LNCS, vol. 8384, pp. 693\u2013703. Springer, Heidelberg (2014). \n                      https:\/\/doi.org\/10.1007\/978-3-642-55224-3_65"},{"issue":"12","key":"37_CR19","doi-asserted-by":"publisher","first-page":"757","DOI":"10.1016\/j.parco.2007.06.001","volume":"33","author":"A Lastovetsky","year":"2007","unstructured":"Lastovetsky, A., Reddy, R.: Data distribution for dense factorization on computers with memory heterogeneity. Parallel Comput. 33(12), 757\u2013779 (2007)","journal-title":"Parallel Comput."},{"issue":"02","key":"37_CR20","doi-asserted-by":"publisher","first-page":"195","DOI":"10.1142\/S0129626411000163","volume":"21","author":"D Clarke","year":"2011","unstructured":"Clarke, D., Lastovetsky, A., Rychkov, V.: Dynamic load balancing of parallel computational iterative routines on highly heterogeneous HPC platforms. Parallel Proces. Lett. 21(02), 195\u2013217 (2011)","journal-title":"Parallel Proces. Lett."},{"key":"37_CR21","unstructured":"AlOnazi, A., Keyes, D., Lastovetsky, A., Rychkov, V.: Design and Optimization of OpenFOAM-based CFD Applications for Hybrid and Heterogeneous HPC Platforms. arXiv preprint \n                      arXiv:1505.07630\n                      \n                     (2015)"},{"key":"37_CR22","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"450","DOI":"10.1007\/978-3-642-29737-3_50","volume-title":"Euro-Par 2011: Parallel Processing Workshops","author":"D Clarke","year":"2012","unstructured":"Clarke, D., Lastovetsky, A., Rychkov, V.: Column-based matrix partitioning for parallel matrix multiplication on heterogeneous processors based on functional performance models. In: Alexander, M., et al. (eds.) Euro-Par 2011. LNCS, vol. 7155, pp. 450\u2013459. Springer, Heidelberg (2012). \n                      https:\/\/doi.org\/10.1007\/978-3-642-29737-3_50"},{"key":"37_CR23","unstructured":"FFTW: Fastest Fourier Transform in the West (2018). \n                      http:\/\/www.fftw.org\/"},{"key":"37_CR24","doi-asserted-by":"publisher","first-page":"1119","DOI":"10.1109\/TPDS.2016.2608824","volume":"28","author":"A Lastovetsky","year":"2017","unstructured":"Lastovetsky, A., Reddy, R.: New model-based methods and algorithms for performance and energy optimization of data parallel applications on homogeneous multicore clusters. IEEE Trans. Parallel Distrib. Syst. 28, 1119\u20131133 (2017)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"37_CR25","doi-asserted-by":"publisher","first-page":"160","DOI":"10.1109\/TC.2017.2742513","volume":"67","author":"R Reddy","year":"2018","unstructured":"Reddy, R., Lastovetsky, A.: Bi-objective optimization of data-parallel applications on homogeneous multicore clusters for performance and energy. IEEE Trans. Comput. 67, 160\u2013177 (2018)","journal-title":"IEEE Trans. Comput."},{"key":"37_CR26","doi-asserted-by":"publisher","first-page":"2176","DOI":"10.1109\/TPDS.2018.2827055","volume":"29","author":"H Khaleghzadeh","year":"2018","unstructured":"Khaleghzadeh, H., Reddy, R., Lastovetsky, A.: A novel data-partitioning algorithm for performance optimization of data-parallel applications on heterogeneous HPC platforms. IEEE Trans. Parallel Distrib. Syst. 29, 2176\u20132190 (2018)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"37_CR27","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1145\/3078811","volume":"50","author":"K O\u2019Brien","year":"2017","unstructured":"O\u2019Brien, K., Petri, I., Reddy, R., Lastovetsky, A., Sakellariou, R.: A survey of power and energy predictive models in HPC systems and applications. ACM Comput. Surv. 50, 37 (2017)","journal-title":"ACM Comput. Surv."},{"key":"37_CR28","first-page":"50","volume":"4","author":"A Shahid","year":"2017","unstructured":"Shahid, A., Fahad, M., Manumachu, R.R., Lastovetsky, A.: Additivity: a selection criterion for performance events for reliable energy predictive modeling. Supercomput. Front. Innov. 4, 50\u201365 (2017)","journal-title":"Supercomput. Front. Innov."}],"container-title":["Lecture Notes in Computer Science","High Performance Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-02465-9_37","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,20]],"date-time":"2019-05-20T05:04:02Z","timestamp":1558328642000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-02465-9_37"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018]]},"ISBN":["9783030024642","9783030024659"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-02465-9_37","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2018]]},"assertion":[{"value":"25 January 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ISC High Performance","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on High Performance Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Frankfurt","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2018","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24 June 2018","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 June 2018","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"33","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"supercomputing2018","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.isc-hpc.com\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}