{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,10]],"date-time":"2026-01-10T07:40:45Z","timestamp":1768030845070,"version":"3.49.0"},"publisher-location":"Cham","reference-count":35,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030186449","type":"print"},{"value":"9783030186456","type":"electronic"}],"license":[{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-030-18645-6_6","type":"book-chapter","created":{"date-parts":[[2019,5,20]],"date-time":"2019-05-20T14:15:30Z","timestamp":1558361730000},"page":"86-105","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":17,"title":["Performance Evaluation and Analysis of Linear Algebra Kernels in the Prototype Tianhe-3 Cluster"],"prefix":"10.1007","author":[{"given":"Xin","family":"You","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hailong","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhongzhi","family":"Luan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Depei","family":"Qian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,4,16]]},"reference":[{"issue":"22","key":"6_CR1","doi-asserted-by":"publisher","first-page":"e4014","DOI":"10.1002\/cpe.4014","volume":"29","author":"JL Bez","year":"2017","unstructured":"Bez, J.L., Bernart, E.E., dos Santos, F.F., Schnorr, L.M., Navaux, P.O.A.: Performance and energy efficiency analysis of HPC physics simulation applications in a cluster of arm processors. Concurrency Comput.: Pract. Experience 29(22), e4014 (2017)","journal-title":"Concurrency Comput.: Pract. Experience"},{"key":"6_CR2","doi-asserted-by":"publisher","DOI":"10.1137\/1.9780898719642","volume-title":"ScaLAPACK Users\u2019 Guide","author":"LS Blackford","year":"1997","unstructured":"Blackford, L.S., et al.: ScaLAPACK Users\u2019 Guide. SIAM, Philadelphia (1997)"},{"key":"6_CR3","unstructured":"Blackmore, C., Ray, O., Eder, K.: Automatically tuning the GCC compiler to optimize the performance of applications running on the ARM cortex-M3. CoRR (2017)"},{"issue":"11","key":"6_CR4","doi-asserted-by":"publisher","first-page":"6201","DOI":"10.1007\/s11227-018-2533-0","volume":"74","author":"N Bock","year":"2018","unstructured":"Bock, N., et al.: The basic matrix library (BML) for quantum chemistry. J. Supercomput. 74(11), 6201\u20136219 (2018)","journal-title":"J. Supercomput."},{"key":"6_CR5","doi-asserted-by":"crossref","unstructured":"Chen, D., Fang, J., Chen, S., Xu, C., Wang, Z.: Optimizing sparse matrix-vector multiplications on an ARMv8-based many-core architecture. Int. J. Parallel Program. 1\u201315 (2018)","DOI":"10.1007\/s10766-018-00625-8"},{"issue":"1","key":"6_CR6","first-page":"1","volume":"38","author":"TA Davis","year":"2011","unstructured":"Davis, T.A., Hu, Y.: The university of florida sparse matrix collection. ACM Trans. Math. Softw. (TOMS) 38(1), 1 (2011)","journal-title":"ACM Trans. Math. Softw. (TOMS)"},{"key":"6_CR7","unstructured":"Arm Developer: Compute Library (2018). https:\/\/developer.arm.com\/technologies\/compute-library"},{"key":"6_CR8","unstructured":"ARM Developer: Arm performance libraries reference guide. ARM Developer (2018)"},{"key":"6_CR9","unstructured":"Dongarra, J.: Report on the TianHe-2a system. Technical report, ICL-UT-17-04, September 2017"},{"key":"6_CR10","unstructured":"FT-2000: Phytium Technology Co., Ltd. (2017). http:\/\/www.phytium.com.cn\/Product\/detail"},{"key":"6_CR11","unstructured":"hir0shim: Open source implentention of distributed SpMV on GitHuB (2015). https:\/\/github.com\/hir0shim\/distributedSpMV"},{"issue":"9","key":"6_CR12","doi-asserted-by":"publisher","first-page":"1073","DOI":"10.1002\/fld.2726","volume":"70","author":"NG Jacobsen","year":"2012","unstructured":"Jacobsen, N.G., Fuhrman, D.R., Freds\u00f8e, J.: A wave generation toolbox for the open-source CFD library: openfoam\u00ae. Int. J. Numer. Methods Fluids 70(9), 1073\u20131088 (2012)","journal-title":"Int. J. Numer. Methods Fluids"},{"key":"6_CR13","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: Advances in Neural Information Processing Systems, pp. 1097\u20131105 (2012)"},{"key":"6_CR14","volume-title":"Google\u2019s PageRank and Beyond: The Science of Search Engine Rankings","author":"AN Langville","year":"2011","unstructured":"Langville, A.N., Meyer, C.D.: Google\u2019s PageRank and Beyond: The Science of Search Engine Rankings. Princeton University Press, Princeton (2011)"},{"key":"6_CR15","doi-asserted-by":"crossref","unstructured":"Liu, W., Vinter, B.: CSR5: an efficient storage format for cross-platform sparse matrix-vector multiplication. In: Proceedings of the 29th ACM on International Conference on Supercomputing, pp. 339\u2013350. ACM (2015)","DOI":"10.1145\/2751205.2751209"},{"key":"6_CR16","doi-asserted-by":"crossref","unstructured":"Liu, X., Smelyanskiy, M., Chow, E., Dubey, P.: Efficient sparse matrix-vector multiplication on x86-based many-core processors. In: Proceedings of the 27th International ACM Conference on International Conference on Supercomputing, pp. 273\u2013282. ACM (2013)","DOI":"10.1145\/2464996.2465013"},{"key":"6_CR17","unstructured":"McCalpin, J.D.: Stream benchmark, vol. 22 (1995). www.cs.virginia.edu\/stream\/ref.html#what"},{"key":"6_CR18","unstructured":"Melnik, D., Belevantsev, A., Plotnikov, D., Lee, S.: A case study: optimizing GCC on ARM for performance of libevas rasterization library. In: Proceedings of GROW (2010)"},{"key":"6_CR19","doi-asserted-by":"crossref","unstructured":"Merrill, D., Garland, M.: Merge-based parallel sparse matrix-vector multiplication. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, p. 58. IEEE Press (2016)","DOI":"10.1109\/SC.2016.57"},{"key":"6_CR20","doi-asserted-by":"crossref","unstructured":"Padoin, E.L., de Oliveira, D.A., Velho, P., Navaux, P.O.: Time-to-solution and energy-to-solution: a comparison between ARM and Xeon. In: 2012 Third Workshop on Applications for Multi-core Architectures (WAMCA), pp. 48\u201353. IEEE (2012)","DOI":"10.1109\/WAMCA.2012.10"},{"key":"6_CR21","unstructured":"Peise, E.: Performance modeling and prediction for dense linear algebra (2017). arXiv preprint arXiv:1706.01341"},{"key":"6_CR22","first-page":"43","volume":"18","author":"S Plimpton","year":"2007","unstructured":"Plimpton, S., Crozier, P., Thompson, A.: Lammps-large-scale atomic\/molecular massively parallel simulator. Sandia Nat. Laboratories 18, 43 (2007)","journal-title":"Sandia Nat. Laboratories"},{"key":"6_CR23","doi-asserted-by":"crossref","unstructured":"Rajovic, N., et al.: The mont-blanc prototype: an alternative approach for HPC systems. In: International Conference for High Performance Computing, Networking, Storage and Analysis, SC 2016, pp. 444\u2013455. IEEE (2016)","DOI":"10.1109\/SC.2016.37"},{"key":"6_CR24","doi-asserted-by":"publisher","first-page":"322","DOI":"10.1016\/j.future.2013.07.013","volume":"36","author":"N Rajovic","year":"2014","unstructured":"Rajovic, N., Rico, A., Puzovic, N., Adeniyi-Jones, C., Ramirez, A.: Tibidabo1: making the case for an ARM-based HPC system. Future Gener. Comput. Syst. 36, 322\u2013334 (2014)","journal-title":"Future Gener. Comput. Syst."},{"issue":"6","key":"6_CR25","doi-asserted-by":"publisher","first-page":"439","DOI":"10.1016\/j.jocs.2013.01.002","volume":"4","author":"N Rajovic","year":"2013","unstructured":"Rajovic, N., Vilanova, L., Villavieja, C., Puzovic, N., Ramirez, A.: The low power architecture approach towards exascale computing. J. Comput. Sci. 4(6), 439\u2013443 (2013)","journal-title":"J. Comput. Sci."},{"key":"6_CR26","unstructured":"Ruiz, D., Mantovani, F., Casas, M., Labarta, J., Spiga, F.: The HPCG benchmark: analysis, shared memory preliminary improvements and evaluation on an arm-based platform (2018)"},{"key":"6_CR27","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/978-3-642-19328-6_1","volume-title":"High Performance Computing for Computational Science \u2013 VECPAR 2010","author":"J Shalf","year":"2011","unstructured":"Shalf, J., Dosanjh, S., Morrison, J.: Exascale computing technology challenges. In: Palma, J.M.L.M., Dayd\u00e9, M., Marques, O., Lopes, J.C. (eds.) VECPAR 2010. LNCS, vol. 6449, pp. 1\u201325. Springer, Heidelberg (2011). https:\/\/doi.org\/10.1007\/978-3-642-19328-6_1"},{"key":"6_CR28","doi-asserted-by":"crossref","unstructured":"Sodani, A.: Knights landing (KNL): 2nd generation Intel\u00ae Xeon Phi processor. In: 2015 IEEE, Hot Chips 27 Symposium (HCS), pp. 1\u201324. IEEE (2015)","DOI":"10.1109\/HOTCHIPS.2015.7477467"},{"key":"6_CR29","unstructured":"Sun, D., Liu, S., Gaudiot, J.L.: Enabling embedded inference engine with arm compute library: a case study (2017). arXiv preprint arXiv:1704.03751"},{"key":"6_CR30","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Ioffe, S., Vanhoucke, V., Alemi, A.A.: Inception-v4, inception-resnet and the impact of residual connections on learning. In: AAAI, vol. 4, p. 12 (2017)","DOI":"10.1609\/aaai.v31i1.11231"},{"key":"6_CR31","doi-asserted-by":"publisher","first-page":"167","DOI":"10.1007\/978-3-319-06486-4_7","volume-title":"High-Performance Computing on the Intel\u00ae Xeon Phi\u2122","author":"E Wang","year":"2014","unstructured":"Wang, E., et al.: Intel math kernel library. In: Wang, E., et al. (eds.) High-Performance Computing on the Intel\u00ae Xeon Phi\u2122, pp. 167\u2013188. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-06486-4_7"},{"key":"6_CR32","doi-asserted-by":"crossref","unstructured":"Williams, S., Oliker, L., Vuduc, R., Shalf, J., Yelick, K., Demmel, J.: Optimization of sparse matrix-vector multiplication on emerging multicore platforms. In: Proceedings of the 2007 ACM\/IEEE Conference on Supercomputing, SC 2007, pp. 1\u201312. IEEE (2007)","DOI":"10.1145\/1362622.1362674"},{"issue":"4","key":"6_CR33","doi-asserted-by":"publisher","first-page":"65","DOI":"10.1145\/1498765.1498785","volume":"52","author":"S Williams","year":"2009","unstructured":"Williams, S., Waterman, A., Patterson, D.: Roofline: an insightful visual performance model for multicore architectures. Commun. ACM 52(4), 65\u201376 (2009)","journal-title":"Commun. ACM"},{"key":"6_CR34","unstructured":"Xianyi, Z., Qian, W., Saar, W.: OpenBLAS: an optimized BLAS library (2016). http:\/\/www.openblas.net\/ . Accessed 12 May 2016"},{"key":"6_CR35","doi-asserted-by":"crossref","unstructured":"Zhang, C.: Mars: a 64-core ARMv8 processor. In: 2015 IEEE Hot Chips 27 Symposium (HCS), pp. 1\u201323. IEEE (2015)","DOI":"10.1109\/HOTCHIPS.2015.7477454"}],"container-title":["Lecture Notes in Computer Science","Supercomputing Frontiers"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-18645-6_6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,18]],"date-time":"2022-09-18T09:04:37Z","timestamp":1663491877000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-18645-6_6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019]]},"ISBN":["9783030186449","9783030186456"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-18645-6_6","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019]]},"assertion":[{"value":"16 April 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"SCFA","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Asian Conference on Supercomputing Frontiers","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Singapore","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Singapore","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2019","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11 March 2019","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 March 2019","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"scfa2019","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.sc-asia.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"OCS","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"33","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"6","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"18% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}}]}}