{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T19:22:46Z","timestamp":1757618566015,"version":"3.44.0"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J. Comput. Sci. Technol."],"published-print":{"date-parts":[[2025,5]]},"DOI":"10.1007\/s11390-023-3489-y","type":"journal-article","created":{"date-parts":[[2025,7,10]],"date-time":"2025-07-10T09:46:07Z","timestamp":1752140767000},"page":"835-854","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Joint-Communication Optimal Matrix Multiplication with Asymmetric Memories"],"prefix":"10.1007","volume":"40","author":[{"given":"Lin","family":"Zhu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiang-Sheng","family":"Hua","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hai","family":"Jin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,7,9]]},"reference":[{"key":"3489_CR1","doi-asserted-by":"publisher","first-page":"111","DOI":"10.1145\/3087556.3087561","volume-title":"Proc. the 29th ACM Symposium on Parallelism in Algorithms and Architectures","author":"E Solomonik","year":"2017","unstructured":"Solomonik E, Ballard G, Demmel J, Hoefler T. A communication-avoiding parallel algorithm for the symmetric eigenvalue problem. In Proc. the 29th ACM Symposium on Parallelism in Algorithms and Architectures, Jul. 2017, pp.111\u2013121. DOI: https:\/\/doi.org\/10.1145\/3087556.3087561."},{"key":"3489_CR2","doi-asserted-by":"publisher","first-page":"307","DOI":"10.1145\/2612669.2612671","volume-title":"Proc. the 26th ACM Symposium on Parallelism in Algorithms and Architectures","author":"E Solomonik","year":"2014","unstructured":"Solomonik E, Carson E, Knight N, Demmel J. Tradeoffs between synchronization, communication, and computation in parallel linear algebra computations. In Proc. the 26th ACM Symposium on Parallelism in Algorithms and Architectures, Jun. 2014, pp.307\u2013318. DOI: https:\/\/doi.org\/10.1145\/2612669.2612671."},{"key":"3489_CR3","doi-asserted-by":"publisher","first-page":"412","DOI":"10.1109\/IPDPS49936.2021.00049","volume-title":"Proc. the 35th IEEE International Parallel and Distributed Processing Symposium","author":"L Ma","year":"2021","unstructured":"Ma L, Solomonik E. Efficient parallel CP decomposition with pairwise perturbation and multi-sweep dimension tree. In Proc. the 35th IEEE International Parallel and Distributed Processing Symposium, May 2021, pp.412\u2013421. DOI: https:\/\/doi.org\/10.1109\/IPDPS49936.2021.00049."},{"issue":"4","key":"3489_CR4","doi-asserted-by":"publisher","first-page":"255","DOI":"10.1002\/(SICI)1096-9128(199704)9:4<255::AID-CPE250>3.0.CO;2-2","volume":"9","author":"R A Van De Geijn","year":"1997","unstructured":"Van De Geijn R A, Watts J. SUMMA: Scalable universal matrix multiplication algorithm. Concurrency: Practice and Experience, 1997, 9(4): 255\u2013274. DOI: https:\/\/doi.org\/10.1002\/(SICI)1096-9128(199704)9:4<255::AID-CPE250>3.0.CO;2-2.","journal-title":"Concurrency: Practice and Experience"},{"key":"3489_CR5","doi-asserted-by":"publisher","first-page":"77","DOI":"10.1145\/2312005.2312021","volume-title":"Proc. the 24th Annual ACM Symposium on Parallelism in Algorithms and Architectures","author":"G Ballard","year":"2012","unstructured":"Ballard G, Demmel J, Holtz O, Lipshitz B, Schwartz O. Brief announcement: Strong scaling of matrix multiplication algorithms and memory-independent communication lower bounds. In Proc. the 24th Annual ACM Symposium on Parallelism in Algorithms and Architectures, Jun. 2012, pp.77\u201379. DOI: https:\/\/doi.org\/10.1145\/2312005.2312021."},{"issue":"3","key":"3489_CR6","doi-asserted-by":"publisher","first-page":"866","DOI":"10.1137\/090769156","volume":"32","author":"G Ballard","year":"2011","unstructured":"Ballard G, Demmel J, Holtz O, Schwartz O. Minimizing communication in numerical linear algebra. SIAM Journal on Matrix Analysis and Applications, 2011, 32(3): 866\u2013901. DOI: https:\/\/doi.org\/10.1137\/090769156.","journal-title":"SIAM Journal on Matrix Analysis and Applications"},{"issue":"9","key":"3489_CR7","doi-asserted-by":"publisher","first-page":"1017","DOI":"10.1016\/j.jpdc.2004.03.021","volume":"64","author":"D Irony","year":"2004","unstructured":"Irony D, Toledo S, Tiskin A. Communication lower bounds for distributed-memory matrix multiplication. Journal of Parallel and Distributed Computing, 2004, 64(9): 1017\u20131026. DOI: https:\/\/doi.org\/10.1016\/j.jpdc.2004.03.021.","journal-title":"Journal of Parallel and Distributed Computing"},{"key":"3489_CR8","doi-asserted-by":"publisher","first-page":"445","DOI":"10.1145\/3490148.3538552","volume-title":"Proc. the 34th ACM Symposium on Parallelism in Algorithms and Architectures","author":"H A Daas","year":"2022","unstructured":"Daas H A, Ballard G, Grigori L, Kumar S, Rouse K. Brief announcement: Tight memory-independent parallel matrix multiplication communication lower bounds. In Proc. the 34th ACM Symposium on Parallelism in Algorithms and Architectures, Jul. 2022, pp.445\u2013448. DOI: https:\/\/doi.org\/10.1145\/3490148.3538552."},{"issue":"5","key":"3489_CR9","doi-asserted-by":"publisher","first-page":"575","DOI":"10.1147\/rd.395.0575","volume":"39","author":"R C Agarwal","year":"1995","unstructured":"Agarwal R C, Balle S M, Gustavson F G, Joshi M, Palkar P. A three-dimensional approach to parallel matrix multiplication. IBM Journal of Research and Development, 1995, 39(5): 575\u2013582. DOI: https:\/\/doi.org\/10.1147\/rd.395.0575.","journal-title":"IBM Journal of Research and Development"},{"key":"3489_CR10","doi-asserted-by":"publisher","first-page":"90","DOI":"10.1007\/978-3-642-23397-5_10","volume-title":"Proc. the 17th International Euro-ParConference on Euro-Par 2011 Parallel Processing","author":"E Solomonik","year":"2011","unstructured":"Solomonik E, Demmel J. Communication-optimal parallel 2.5D matrix multiplication and LU factorization algorithms. In Proc. the 17th International Euro-ParConference on Euro-Par 2011 Parallel Processing, Aug. 29\u2013Sept. 2, 2011, pp.90\u2013109. DOI: https:\/\/doi.org\/10.1007\/978-3-642-23397-5_10."},{"key":"3489_CR11","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00033","volume-title":"Proc. the 2022 International Conference for High Performance Computing, Networking, Storage and Analysis","author":"H Huang","year":"2022","unstructured":"Huang H, Chow E. CA3DMM: A new algorithm based on a unified view of parallel matrix multiplication. In Proc. the 2022 International Conference for High Performance Computing, Networking, Storage and Analysis, Nov. 2022. DOI: https:\/\/doi.org\/10.1109\/SC41404.2022.00033."},{"key":"3489_CR12","doi-asserted-by":"publisher","first-page":"1077","DOI":"10.1145\/3373376.3378515","volume-title":"Proc. the 25th International Conference on Architectural Support for Programming Languages and Operating Systems","author":"Y Chen","year":"2020","unstructured":"Chen Y, Lu Y, Yang F, Wang Q, Wang Y, Shu J. Flat-Store: An efficient log-structured key-value storage engine for persistent memory. In Proc. the 25th International Conference on Architectural Support for Programming Languages and Operating Systems, Mar. 2020, pp.1077\u20131091. DOI: https:\/\/doi.org\/10.1145\/3373376.3378515."},{"key":"3489_CR13","first-page":"523","volume-title":"Proc. the 2021 USENIX Annual Technical Conference","author":"X Wei","year":"2021","unstructured":"Wei X, Xie X, Chen R, Chen H, Zang B. Characterizing and optimizing remote persistent memory with RDMA and NVM. In Proc. the 2021 USENIX Annual Technical Conference, Jul. 2021, pp.523\u2013536."},{"key":"3489_CR14","doi-asserted-by":"publisher","first-page":"2765","DOI":"10.1145\/3548606.3560568","volume-title":"Proc. the 2022 ACM SIGSAC Conference on Computer and Communications Security","author":"K Taranov","year":"2022","unstructured":"Taranov K, Rothenberger B, De Sensi D, Perrig A, Hoefler T. NeVerMore: Exploiting RDMA mistakes in NVMeoF storage applications. In Proc. the 2022 ACM SIGSAC Conference on Computer and Communications Security, Nov. 2022, pp.2765\u20132778. DOI: https:\/\/doi.org\/10.1145\/3548606.3560568."},{"key":"3489_CR15","series-title":"Technical Report UCB\/EECS-2015-163","volume-title":"Write-avoiding algorithms","author":"E Carson","year":"2015","unstructured":"Carson E, Demmel J, Grigori L, Knight N, Koanantakool P, Schwartz O, Simhadri H V. Write-avoiding algorithms. Technical Report UCB\/EECS-2015-163, EECS Department, University of California, 2015. http:\/\/www2.eecs.berkeley.edu\/Pubs\/TechRpts\/2015\/EECS-2015-163.html, May 2025."},{"key":"3489_CR16","series-title":"Ph. D. Thesis","volume-title":"Write-efficient algorithms","author":"Y Gu","year":"2018","unstructured":"Gu Y. Write-efficient algorithms [Ph. D. Thesis]. Carnegie Mellon University, Pittsburgh, 2018."},{"key":"3489_CR17","doi-asserted-by":"publisher","first-page":"648","DOI":"10.1109\/IPDPS.2016.114","volume-title":"Proc. the 2016 IEEE International Parallel and Distributed Processing Symposium","author":"E C Carson","year":"2016","unstructured":"Carson E C, Demmel J, Grigori L, Knight N, Koanantakool P, Schwartz O, Simhadri H V. Write-avoiding algorithms. In Proc. the 2016 IEEE International Parallel and Distributed Processing Symposium, May 2016, pp.648\u2013658. DOI: https:\/\/doi.org\/10.1109\/IPDPS.2016.114."},{"key":"3489_CR18","doi-asserted-by":"publisher","first-page":"261","DOI":"10.1109\/IPDPS.2013.80","volume-title":"Proc. the 27th IEEE International Symposium on Parallel and Distributed Processing","author":"J Demmel","year":"2013","unstructured":"Demmel J, Eliahu D, Fox A, Kamil S, Lipshitz B, Schwartz O, Spillinger O. Communication-optimal parallel recursive rectangular matrix multiplication. In Proc. the 27th IEEE International Symposium on Parallel and Distributed Processing, May 2013, pp.261\u2013272. DOI: https:\/\/doi.org\/10.1109\/IPDPS.2013.80."},{"key":"3489_CR19","doi-asserted-by":"publisher","first-page":"285","DOI":"10.1109\/SFFCS.1999.814600","volume-title":"Proc. the 40th Annual Symposium on Foundations of Computer Science","author":"M Frigo","year":"1999","unstructured":"Frigo M, Leiserson C E, Prokop H, Ramachandran S. Cacheoblivious algorithms. In Proc. the 40th Annual Symposium on Foundations of Computer Science, Oct. 1999, pp.285\u2013298. DOI: https:\/\/doi.org\/10.1109\/SFFCS.1999.814600."},{"key":"3489_CR20","doi-asserted-by":"publisher","first-page":"193","DOI":"10.1145\/2312005.2312044","volume-title":"Proc. the 24th Annual ACM Symposium on Parallelism in Algorithms and Architectures","author":"G Ballard","year":"2012","unstructured":"Ballard G, Demmel J, Holtz O, Lipshitz B, Schwartz O. Communication-optimal parallel algorithm for Strassen\u2019s matrix multiplication. In Proc. the 24th Annual ACM Symposium on Parallelism in Algorithms and Architectures, Jun. 2012, pp.193\u2013204. DOI: https:\/\/doi.org\/10.1145\/2312005.2312044."},{"key":"3489_CR21","doi-asserted-by":"publisher","first-page":"222","DOI":"10.1145\/2486159.2486196","volume-title":"Proc. the 25th Annual ACM Symposium on Parallelism in Algorithms and Architectures","author":"G Ballard","year":"2013","unstructured":"Ballard G, Bulu\u00e7 A, Demmel J, Grigori L, Lipshitz B, Schwartz O, Toledo S. Communication optimal parallel multiplication of sparse random matrices. In Proc. the 25th Annual ACM Symposium on Parallelism in Algorithms and Architectures, Jul. 2013, pp.222\u2013231. DOI: https:\/\/doi.org\/10.1145\/2486159.2486196."},{"key":"3489_CR22","doi-asserted-by":"publisher","first-page":"498","DOI":"10.1145\/3582016.3582055","volume-title":"Proc. the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems","author":"C Ruan","year":"2023","unstructured":"Ruan C, Zhang Y, Bi C, Ma X, Chen H, Li F, Yang X, Li C, Aboulnaga A, Xu Y. Persistent memory disaggregation for cloud-native relational databases. In Proc. the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Mar. 2023, pp.498\u2013512. DOI: https:\/\/doi.org\/10.1145\/3582016.3582055."},{"key":"3489_CR23","doi-asserted-by":"publisher","first-page":"366","DOI":"10.1109\/DCS.1988.12538","volume-title":"Proc. the 8th International Conference on Distributed","author":"C Q Yang","year":"1988","unstructured":"Yang C Q, Miller B P. Critical path analysis for the execution of parallel and distributed programs. In Proc. the 8th International Conference on Distributed, Jun. 1988, pp.366\u2013373. DOI: https:\/\/doi.org\/10.1109\/DCS.1988.12538."},{"issue":"11","key":"3489_CR24","doi-asserted-by":"publisher","first-page":"212101","DOI":"10.1007\/s11432-020-2931-x","volume":"64","author":"Q S Hua","year":"2021","unstructured":"Hua Q S, Qian L, Yu D, Shi X, Jin H. A nearly optimal distributed algorithm for computing the weighted girth. Science China Information Sciences, 2021, 64(11): 212101. DOI: https:\/\/doi.org\/10.1007\/s11432-020-2931-x.","journal-title":"Science China Information Sciences"},{"issue":"5","key":"3489_CR25","doi-asserted-by":"publisher","first-page":"152101","DOI":"10.1007\/s11432-020-2996-2","volume":"65","author":"L Jia","year":"2022","unstructured":"Jia L, Hua Q S, Fan H, Wang Q, Jin H. Efficient distributed algorithms for holistic aggregation functions on random regular graphs. Science China Information Sciences, 2022, 65(5): 152101. DOI: https:\/\/doi.org\/10.1007\/s11432-020-2996-2.","journal-title":"Science China Information Sciences"},{"issue":"13","key":"3489_CR26","doi-asserted-by":"publisher","first-page":"1749","DOI":"10.1002\/cpe.1206","volume":"19","author":"E Chan","year":"2007","unstructured":"Chan E, Heimlich M, Purkayastha A, van De Geijn R. Collective communication: Theory, practice, and experience. Concurrency and Computation: Practice and Experience, 2007, 19(13): 1749\u20131783. DOI: https:\/\/doi.org\/10.1002\/cpe.1206.","journal-title":"Concurrency and Computation: Practice and Experience"},{"issue":"1","key":"3489_CR27","doi-asserted-by":"publisher","first-page":"49","DOI":"10.1177\/1094342005051521","volume":"19","author":"R Thakur","year":"2005","unstructured":"Thakur R, Rabenseifner R, Gropp W. Optimization of collective communication operations in MPICH. The International Journal of High Performance Computing Applications, 2005, 19(1): 49\u201366. DOI: https:\/\/doi.org\/10.1177\/1094342005051521.","journal-title":"The International Journal of High Performance Computing Applications"},{"key":"3489_CR28","doi-asserted-by":"publisher","first-page":"89","DOI":"10.1109\/IPDPS.2019.00020","volume-title":"Proc. the 2019 IEEE International Parallel and Distributed Processing Symposium","author":"E Hutter","year":"2019","unstructured":"Hutter E, Solomonik E. Communication-avoiding CholeskyQR2 for rectangular matrices. In Proc. the 2019 IEEE International Parallel and Distributed Processing Symposium, May 2019, pp.89\u2013100. DOI: https:\/\/doi.org\/10.1109\/IPDPS.2019.00020."},{"key":"3489_CR29","doi-asserted-by":"publisher","first-page":"326","DOI":"10.1145\/800076.802486","volume-title":"Proc. the 13th Annual ACM Symposium on Theory of Computing","author":"J W Hong","year":"1981","unstructured":"Hong J W, Kung H T. I\/O complexity: The red-blue pebble game. In Proc. the 13th Annual ACM Symposium on Theory of Computing, May 1981, pp.326\u2013333. DOI: https:\/\/doi.org\/10.1145\/800076.802486."},{"issue":"10","key":"3489_CR30","doi-asserted-by":"publisher","first-page":"961","DOI":"10.1090\/S0002-9904-1949-09320-5","volume":"55","author":"L H Loomis","year":"1949","unstructured":"Loomis L H, Whitney H. An inequality related to the isoperimetric inequality. Bulletin of the American Mathematical Society, 1949, 55(10): 961\u2013962. DOI: https:\/\/doi.org\/10.1090\/S0002-9904-1949-09320-5.","journal-title":"Bulletin of the American Mathematical Society"},{"key":"3489_CR31","doi-asserted-by":"publisher","first-page":"120","DOI":"10.1109\/FMPC.1992.234898","volume-title":"Proc. the 4th Symposium on the Frontiers of Massively Parallel Computation","author":"J Choi","year":"1992","unstructured":"Choi J, Dongarra J J, Pozo R, Walker D W. ScaLA-PACK: A scalable linear algebra library for distributed memory concurrent computers. In Proc. the 4th Symposium on the Frontiers of Massively Parallel Computation, Oct. 1992, pp.120\u2013127. DOI: https:\/\/doi.org\/10.1109\/FMPC.1992.234898."},{"key":"3489_CR32","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356181","volume-title":"Proc. the 2019 International Conference for High Performance Computing, Networking, Storage and Analysis","author":"G Kwasniewski","year":"2019","unstructured":"Kwasniewski G, Kabi\u0107 M, Besta M, VandeVondele J, Solc\u00e0 R, Hoefler T. Red-blue pebbling revisited: Near optimal parallel matrix-matrix multiplication. In Proc. the 2019 International Conference for High Performance Computing, Networking, Storage and Analysis, Nov. 2019, Article No. 24. DOI: https:\/\/doi.org\/10.1145\/3295500.3356181."},{"key":"3489_CR33","doi-asserted-by":"publisher","first-page":"588","DOI":"10.1145\/3575693.3575722","volume-title":"Proc. the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems","author":"Y Song","year":"2023","unstructured":"Song Y, Kim W H, Monga S K, Min C, Eom Y I. Prism: Optimizing key-value store for modern heterogeneous storage devices. In Proc. the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Mar. 2023, pp.588\u2013602. DOI: https:\/\/doi.org\/10.1145\/3575693.3575722."},{"key":"3489_CR34","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2008.4536305","volume-title":"Proc. the 2008 IEEE International Symposium on Parallel and Distributed Processing","author":"J Demmel","year":"2008","unstructured":"Demmel J, Hoemmen M, Mohiyuddin M, Yelick K. Avoiding communication in sparse matrix computations. In Proc. the 2008 IEEE International Symposium on Parallel and Distributed Processing, Apr. 2008. DOI: https:\/\/doi.org\/10.1109\/IPDPS.2008.4536305."}],"container-title":["Journal of Computer Science and Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11390-023-3489-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11390-023-3489-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11390-023-3489-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,7]],"date-time":"2025-09-07T02:16:51Z","timestamp":1757211411000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11390-023-3489-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5]]},"references-count":34,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2025,5]]}},"alternative-id":["3489"],"URL":"https:\/\/doi.org\/10.1007\/s11390-023-3489-y","relation":{},"ISSN":["1000-9000","1860-4749"],"issn-type":[{"type":"print","value":"1000-9000"},{"type":"electronic","value":"1860-4749"}],"subject":[],"published":{"date-parts":[[2025,5]]},"assertion":[{"value":"12 June 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 January 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 July 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Conflict of Interest The authors declare that they have no conflict of interest.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics"}}]}}