{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:55:19Z","timestamp":1776930919148,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":35,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U2333211"],"award-info":[{"award-number":["U2333211"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"University of Electronic Science and Technology of China startup grant","award":["1098531023601465"],"award-info":[{"award-number":["1098531023601465"]}]},{"name":"CCF-SUMA foundation grant","award":["CCF-GH OF 2024002."],"award-info":[{"award-number":["CCF-GH OF 2024002."]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,16]]},"DOI":"10.1145\/3712285.3759770","type":"proceedings-article","created":{"date-parts":[[2025,11,12]],"date-time":"2025-11-12T16:04:47Z","timestamp":1762963487000},"page":"1830-1844","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Rethinking Back Transformation in 2-stage Eigenvalue Decomposition on Heterogeneous Architectures"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-0035-2323","authenticated-orcid":false,"given":"Hansheng","family":"Wang","sequence":"first","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2857-1571","authenticated-orcid":false,"given":"Dajun","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-0550-2662","authenticated-orcid":false,"given":"Gaoyuan","family":"Zou","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2481-0318","authenticated-orcid":false,"given":"Lu","family":"Shi","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2675-2895","authenticated-orcid":false,"given":"Xu","family":"Jiang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7659-1631","authenticated-orcid":false,"given":"Xi","family":"Wu","sequence":"additional","affiliation":[{"name":"Chengdu University of Information Technology, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7721-7422","authenticated-orcid":false,"given":"Hancong","family":"Duan","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9525-1659","authenticated-orcid":false,"given":"Shaoshuai","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,11,15]]},"reference":[{"key":"e_1_3_3_3_2_2","doi-asserted-by":"crossref","unstructured":"Herv\u00e9 Abdi and Lynne\u00a0J Williams. 2010. Principal component analysis. Wiley interdisciplinary reviews: computational statistics 2 4 (2010) 433\u2013459.","DOI":"10.1002\/wics.101"},{"key":"e_1_3_3_3_3_2","doi-asserted-by":"publisher","DOI":"10.5555\/323215"},{"key":"e_1_3_3_3_4_2","doi-asserted-by":"crossref","unstructured":"Christian Bischof and Charles Van\u00a0Loan. 1987. The WY representation for products of Householder matrices. SIAM J. Sci. Statist. Comput. 8 1 (1987) s2\u2013s13.","DOI":"10.1137\/0908009"},{"key":"e_1_3_3_3_5_2","doi-asserted-by":"crossref","unstructured":"Christian\u00a0H. Bischof Xiaobai Sun and Bruno Lang. 1994. Parallel tridiagonalization through two-step band reduction. Proceedings of IEEE Scalable High Performance Computing Conference (1994) 23\u201327.","DOI":"10.1109\/SHPCC.1994.296622"},{"key":"e_1_3_3_3_6_2","doi-asserted-by":"crossref","unstructured":"H Chermette. 1998. Density functional theory: a powerful tool for theoretical studies in coordination chemistry. Coordination chemistry reviews 178 (1998) 699\u2013721.","DOI":"10.1016\/S0010-8545(98)00179-9"},{"key":"e_1_3_3_3_7_2","doi-asserted-by":"crossref","unstructured":"Jack Choquette. 2023. Nvidia hopper h100 gpu: Scaling performance. IEEE Micro (2023).","DOI":"10.1109\/MM.2023.3256796"},{"key":"e_1_3_3_3_8_2","doi-asserted-by":"crossref","unstructured":"Jack Choquette Wishwesh Gandhi Olivier Giroux Nick Stam and Ronny Krashinsky. 2021. Nvidia a100 tensor core gpu: Performance and innovation. IEEE Micro 41 2 (2021) 29\u201335.","DOI":"10.1109\/MM.2021.3061394"},{"key":"e_1_3_3_3_9_2","doi-asserted-by":"crossref","unstructured":"Carles Curutchet and Benedetta Mennucci. 2017. Quantum chemical studies of light harvesting. Chemical reviews 117 2 (2017) 294\u2013343.","DOI":"10.1021\/acs.chemrev.5b00700"},{"key":"e_1_3_3_3_10_2","doi-asserted-by":"crossref","unstructured":"Jack Dongarra Mark Gates Azzam Haidar Jakub Kurzak Piotr Luszczek Panruo Wu Ichitaro Yamazaki Asim YarKhan Maksims Abalenkovs Negin Bagherpour et\u00a0al. 2019. PLASMA: Parallel linear algebra software for multicore using OpenMP. ACM Transactions on Mathematical Software (TOMS) 45 2 (2019) 1\u201335.","DOI":"10.1145\/3264491"},{"key":"e_1_3_3_3_11_2","doi-asserted-by":"crossref","unstructured":"Jack\u00a0J Dongarra Danny\u00a0C Sorensen and Sven\u00a0J Hammarling. 1989. Block reduction of matrices to condensed forms for eigenvalue computations. J. Comput. Appl. Math. 27 1-2 (1989) 215\u2013227.","DOI":"10.1016\/0377-0427(89)90367-1"},{"key":"e_1_3_3_3_12_2","unstructured":"Sebastian Gant. [n.d.]. Chasing the Bulge. ([n. d.])."},{"key":"e_1_3_3_3_13_2","doi-asserted-by":"crossref","unstructured":"Gene\u00a0H Golub and Henk\u00a0A Van\u00a0der Vorst. 2000. Eigenvalue computation in the 20th century. J. Comput. Appl. Math. 123 1-2 (2000) 35\u201365.","DOI":"10.1016\/S0377-0427(00)00413-1"},{"key":"e_1_3_3_3_14_2","doi-asserted-by":"crossref","unstructured":"Roger Grimes Henry Krakauer John Lewis Horst Simon and Su-Hai Wei. 1987. The solution of large dense generalized eigenvalue problems on the Cray X-MP\/24 with SSD. J. Comput. Phys. 69 2 (1987) 471\u2013481.","DOI":"10.1016\/0021-9991(87)90178-1"},{"key":"e_1_3_3_3_15_2","doi-asserted-by":"crossref","unstructured":"Ming Gu and Stanley\u00a0C Eisenstat. 1995. A divide-and-conquer algorithm for the symmetric tridiagonal eigenproblem. SIAM J. Matrix Anal. Appl. 16 1 (1995) 172\u2013191.","DOI":"10.1137\/S0895479892241287"},{"key":"e_1_3_3_3_16_2","unstructured":"Ga\u00ebl Guennebaud Benoit Jacob et\u00a0al. 2010. Eigen. URl: http:\/\/eigen. tuxfamily. org 3 1 (2010)."},{"key":"e_1_3_3_3_17_2","first-page":"1842","volume-title":"International Conference on Machine Learning","author":"Gupta Vineet","year":"2018","unstructured":"Vineet Gupta, Tomer Koren, and Yoram Singer. 2018. Shampoo: Preconditioned stochastic tensor optimization. In International Conference on Machine Learning. PMLR, 1842\u20131850."},{"key":"e_1_3_3_3_18_2","doi-asserted-by":"publisher","DOI":"10.1145\/2063384.2063394"},{"key":"e_1_3_3_3_19_2","doi-asserted-by":"crossref","unstructured":"Azzam Haidar Stanimire Tomov Jack Dongarra Raffaele Solca and Thomas Schulthess. 2014. A novel hybrid CPU\u2013GPU generalized eigensolver for electronic structure calculations based on fine-grained memory aware tasks. The International journal of high performance computing applications 28 2 (2014) 196\u2013209.","DOI":"10.1177\/1094342013502097"},{"key":"e_1_3_3_3_20_2","doi-asserted-by":"crossref","unstructured":"Jiajun Li Denis Golez Giacomo Mazza Andrew\u00a0J Millis Antoine Georges and Martin Eckstein. 2020. Electromagnetic coupling in tight-binding models for strongly correlated light and matter. Physical Review B 101 20 (2020) 205140.","DOI":"10.1103\/PhysRevB.101.205140"},{"key":"e_1_3_3_3_21_2","doi-asserted-by":"publisher","DOI":"10.3233\/978-1-61499-041-3-397"},{"key":"e_1_3_3_3_22_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2011.91"},{"key":"e_1_3_3_3_23_2","doi-asserted-by":"crossref","unstructured":"Andreas Marek Volker Blum Rainer Johanni Ville Havu Bruno Lang Thomas Auckenthaler Alexander Heinecke Hans-Joachim Bungartz and Hermann Lederer. 2014. The ELPA library: scalable parallel eigenvalue solutions for electronic structure theory and computational science. Journal of Physics: Condensed Matter 26 21 (2014) 213201.","DOI":"10.1088\/0953-8984\/26\/21\/213201"},{"key":"e_1_3_3_3_24_2","doi-asserted-by":"crossref","unstructured":"Matt Probert. 2011. Electronic Structure: Basic Theory and Practical Methods by Richard M. Martin: Scope: graduate level textbook. Level: theoretical materials scientists\/condensed matter physicists\/computational chemists.","DOI":"10.1080\/00107514.2010.509989"},{"key":"e_1_3_3_3_25_2","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-28555-5_25"},{"key":"e_1_3_3_3_26_2","doi-asserted-by":"crossref","unstructured":"Robert Schreiber and Charles Van\u00a0Loan. 1989. A storage-efficient WY representation for products of Householder transformations. SIAM J. Sci. Statist. Comput. 10 1 (1989) 53\u201357.","DOI":"10.1137\/0910005"},{"key":"e_1_3_3_3_27_2","doi-asserted-by":"publisher","DOI":"10.1109\/BigData47090.2019.9006315"},{"key":"e_1_3_3_3_28_2","doi-asserted-by":"crossref","unstructured":"ECG Sudarshan CB Chiu and Vittorio Gorini. 1978. Decaying states as complex energy eigenvectors in generalized quantum mechanics. Physical Review D 18 8 (1978) 2914.","DOI":"10.1103\/PhysRevD.18.2914"},{"key":"e_1_3_3_3_29_2","doi-asserted-by":"crossref","unstructured":"Dalal Sukkari Hatem Ltaief and David Keyes. 2016. A high performance QDWH-SVD solver using hardware accelerators. ACM Transactions on Mathematical Software (TOMS) 43 1 (2016) 1\u201325.","DOI":"10.1145\/2894747"},{"key":"e_1_3_3_3_30_2","doi-asserted-by":"crossref","unstructured":"Fran\u00e7oise Tisseur and Jack Dongarra. 1999. A parallel divide and conquer algorithm for the symmetric eigenvalue problem on distributed memory architectures. SIAM Journal on Scientific Computing 20 6 (1999) 2223\u20132236.","DOI":"10.1137\/S1064827598336951"},{"key":"e_1_3_3_3_31_2","unstructured":"Stanimire Tomov Rajib Nath Peng Du and Jack Dongarra. 2011. MAGMA Users\u2019 Guide. ICL UTK (November 2009) (2011)."},{"key":"e_1_3_3_3_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/3710848.3710894"},{"key":"e_1_3_3_3_33_2","doi-asserted-by":"crossref","unstructured":"David Watkins and Ludwig Elsner. 1994. Theory of decomposition and bulge-chasing algorithms for the generalized eigenvalue problem. SIAM Journal on matrix analysis and applications 15 3 (1994) 943\u2013967.","DOI":"10.1137\/S089547989122377X"},{"key":"e_1_3_3_3_34_2","doi-asserted-by":"crossref","unstructured":"David\u00a0S Watkins. 1982. Understanding the QR algorithm. SIAM review 24 4 (1982) 427\u2013440.","DOI":"10.1137\/1024100"},{"key":"e_1_3_3_3_35_2","doi-asserted-by":"crossref","unstructured":"Victor Wen-zhe Yu Jonathan Moussa Pavel Kus Andreas Marek Peter Messmer Mina Yoon Hermann Lederer and Volker Blum. 2021. GPU-acceleration of the ELPA2 distributed eigensolver for dense symmetric and hermitian eigenproblems. Computer Physics Communications 262 (2021) 107808.","DOI":"10.1016\/j.cpc.2020.107808"},{"key":"e_1_3_3_3_36_2","doi-asserted-by":"publisher","DOI":"10.1145\/3392717.3392770"}],"event":{"name":"SC '25: The International Conference for High Performance Computing, Networking, Storage and Analysis","location":"St. Louis MO USA","acronym":"SC '25","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3712285.3759770","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T18:42:20Z","timestamp":1773254540000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3712285.3759770"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,15]]},"references-count":35,"alternative-id":["10.1145\/3712285.3759770","10.1145\/3712285"],"URL":"https:\/\/doi.org\/10.1145\/3712285.3759770","relation":{},"subject":[],"published":{"date-parts":[[2025,11,15]]},"assertion":[{"value":"2025-11-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}