{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T16:15:44Z","timestamp":1773591344351,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","funder":[{"name":"National Key Research and Development Program of China","award":["2025YFB3003202"],"award-info":[{"award-number":["2025YFB3003202"]}]},{"name":"Pilot Project for Integrated Innovation of Science Education and Industry of Qilu University of Technology","award":["2024GH24"],"award-info":[{"award-number":["2024GH24"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,8,27]]},"DOI":"10.1145\/3774949.3774961","type":"proceedings-article","created":{"date-parts":[[2026,1,24]],"date-time":"2026-01-24T16:04:14Z","timestamp":1769270654000},"page":"81-86","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["A Heterogeneous Parallel Optimization Algorithm for Batched Matrix Multiplications on the SW26010-pro Processor"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-0332-7687","authenticated-orcid":false,"given":"Long","family":"Zhang","sequence":"first","affiliation":[{"name":"Key Laboratory of Computing Power Network and Information Security, Ministry of Education, Shandong Computer Science Center (National Supercomputer Center in Jinan), Qilu University of Technology (Shandong Academy of Sciences), Jinan, Shandong, China and Shandong Provincial Key Laboratory of Computing Power Internet and Service Computing, Shandong Fundamental Research Center for Computer Science, Jinan, China, Shandong Provincial Key Laboratory of Computing Power Internet and Service Computing, Shandong Fundamental Research Center for Computer Science, Jinan, China, Jinan, Shandong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9930-4802","authenticated-orcid":false,"given":"Min","family":"Tian","sequence":"additional","affiliation":[{"name":"Key Laboratory of Computing Power Network and Information Security, Ministry of Education, Shandong Computer Science Center (National Supercomputer Center in Jinan), Qilu University of Technology (Shandong Academy of Sciences), Jinan, Shandong, China and Shandong Provincial Key Laboratory of Computing Power Internet and Service Computing, Shandong Fundamental Research Center for Computer Science, Jinan, China, Shandong Provincial Key Laboratory of Computing Power Internet and Service Computing, Shandong Fundamental Research Center for Computer Science, Jinan, China, Jinan, Shandong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7064-1869","authenticated-orcid":false,"given":"Shui","family":"Nai","sequence":"additional","affiliation":[{"name":"Key Laboratory of Computing Power Network and Information Security, Ministry of Education, Shandong Computer Science Center (National Supercomputer Center in Jinan), Qilu University of Technology (Shandong Academy of Sciences), Jinan, Shandong, China and Shandong Provincial Key Laboratory of Computing Power Internet and Service Computing, Shandong Fundamental Research Center for Computer Science, Jinan, China, Shandong Provincial Key Laboratory of Computing Power Internet and Service Computing, Shandong Fundamental Research Center for Computer Science, Jinan, China, Jinan, Shandong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-5030-5736","authenticated-orcid":false,"given":"Xinyu","family":"Guo","sequence":"additional","affiliation":[{"name":"Key Laboratory of Computing Power Network and Information Security, Ministry of Education, Shandong Computer Science Center (National Supercomputer Center in Jinan), Qilu University of Technology (Shandong Academy of Sciences), Jinan, Shandong, China and Shandong Provincial Key Laboratory of Computing Power Internet and Service Computing, Shandong Fundamental Research Center for Computer Science, Jinan, China, Shandong Provincial Key Laboratory of Computing Power Internet and Service Computing, Shandong Fundamental Research Center for Computer Science, Jinan, China, Jinan, Shandong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,1,24]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","unstructured":"Martin Abadi Ashish Agarwal Paul Barham Eugene Brevdo Zhifeng Chen Craig Citro Greg\u00a0S. Corrado Andy Davis Jeffrey Dean Matthieu Devin et\u00a0al. 2015. TensorFlow: Large-scale machine learning on heterogeneous systems. Software available from tensorflow.org 1 2 (2015). 10.1177\/1094342010385729","DOI":"10.1177\/1094342010385729"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-41321-1_2"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1088\/1742-6596\/180\/1\/012037"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCC\/SmartCity\/DSS.2019.00203"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"crossref","unstructured":"Jack Dongarra Sven Hammarling Nicholas\u00a0J Higham Samuel\u00a0D Relton Pedro Valero-Lara and Mawussi Zounon. 2017. The design and performance of batched BLAS on modern high-performance computing systems. Procedia Computer Science 108 (2017) 495\u2013504.","DOI":"10.1016\/j.procs.2017.05.138"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.5555\/3433701.3433723"},{"key":"e_1_3_3_1_8_2","unstructured":"Intel Corporation. 2019. https:\/\/software.intel.com\/en-us\/intel-mkl"},{"key":"e_1_3_3_1_9_2","unstructured":"Xu J Huang Y Guo S et\u00a0al. 2015. Testing platform for floating mathematical function libraries. Journal of Software 26 6 (2015) 1306\u20131321."},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"crossref","unstructured":"Chetan Jhurani and Paul Mullowney. 2015. A GEMM interface and implementation on NVIDIA GPUs for multiple small matrices. J. Parallel and Distrib. Comput. 75 (2015) 133\u2013140.","DOI":"10.1016\/j.jpdc.2014.09.003"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2017.51"},{"key":"e_1_3_3_1_12_2","unstructured":"William Kahan. 1996. IEEE standard 754 for binary floating-point arithmetic. Lecture Notes on the Status of IEEE 754 94720-1776 (1996) 11."},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1093\/acprof:oso\/9780198528692.001.0001"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"crossref","unstructured":"Mingfan Li Junshi Chen Qian Xiao Fei Wang Qingcai Jiang Xuncheng Zhao Rongfen Lin Hong An Xiao Liang and Lixin He. 2022. Bridging the gap between deep learning and frustrated quantum spin system for extreme-scale simulations on new generation of Sunway supercomputer. IEEE Transactions on Parallel and Distributed Systems 33 11 (2022) 2846\u20132859.","DOI":"10.1109\/TPDS.2022.3145163"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"crossref","unstructured":"Fangfang Liu Wenjing Ma Yuwen Zhao Daokun Chen Yi Hu Qinglin Lu WanWang Yin Xinhui Yuan Lijuan Jiang Hao Yan et\u00a0al. 2023. xmath2. 0: a high-performance extended math library for sw26010-pro many-core processor. CCF Transactions on High Performance Computing 5 1 (2023) 56\u201371.","DOI":"10.1007\/s42514-022-00126-8"},{"key":"e_1_3_3_1_16_2","unstructured":"MAGMA Project. 2019. http:\/\/icl.cs.utk.edu\/magma\/"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"crossref","unstructured":"Ian Masliah Ahmad Abdelfattah Azzam Haidar Stanimire Tomov Marc Baboulin Jo\u00ebl Falcou and Jack Dongarra. 2019. Algorithms and optimization techniques for high-performance matrix-matrix multiplications of very small matrices. Parallel Comput. 81 (2019) 1\u201321.","DOI":"10.1016\/j.parco.2018.10.003"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.23919\/MIPRO55190.2022.9803591"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"crossref","unstructured":"Jose\u00a0M Molero Ester\u00a0M Garz\u00f3n Inmaculada Garc\u00eda Enrique\u00a0S Quintana-Ort\u00ed and Antonio Plaza. 2014. Efficient implementation of hyperspectral anomaly detection techniques on GPUs and multicore processors. IEEE Journal of selected topics in applied earth observations and remote sensing 7 6 (2014) 2256\u20132266.","DOI":"10.1109\/JSTARS.2014.2328614"},{"key":"e_1_3_3_1_20_2","unstructured":"MPI Forum. 2021. https:\/\/www.mpi-forum.org\/docs\/mpi-4.0\/mpi40-report.pdf"},{"key":"e_1_3_3_1_21_2","unstructured":"NVIDIA Corporation. 2019. https:\/\/docs.nvidia.com\/cuda\/cublas\/"},{"key":"e_1_3_3_1_22_2","unstructured":"OpenBLAS Project. 2023. https:\/\/www.openblas.net\/documentation.html"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.23919\/ISC.2024.10528930"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","unstructured":"Liu SF Zhao YH Huang RF Yu TY and Zhang XY. 2023. Effective Implementation of Matrix Inversion Based on Batched LU Decomposition on GPU. Ruan Jian Xue Bao\/Journal of Software 34 11 (2023) 4952\u20134972. 10.13328\/j.cnki.jos.006727","DOI":"10.13328\/j.cnki.jos.006727"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISPDC.2019.00021"},{"key":"e_1_3_3_1_26_2","unstructured":"Fan Zhang Chen Hu Qiang Yin and Wei Hu. 2017. A GPU based memory optimized parallel method for FFT implementation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1707.07263 (2017)."}],"event":{"name":"HP3C 2025: 2025 9th International Conference on High Performance Compilation, Computing and Communications (HP3C)","location":"Jinan , China","acronym":"HP3C 2025"},"container-title":["Proceedings of the 2025 9th International Conference on High Performance Compilation, Computing and Communications"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774949.3774961","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T15:28:36Z","timestamp":1773588516000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774949.3774961"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,27]]},"references-count":25,"alternative-id":["10.1145\/3774949.3774961","10.1145\/3774949"],"URL":"https:\/\/doi.org\/10.1145\/3774949.3774961","relation":{},"subject":[],"published":{"date-parts":[[2025,8,27]]},"assertion":[{"value":"2026-01-24","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}