{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T03:58:35Z","timestamp":1781668715864,"version":"3.54.5"},"publisher-location":"Cham","reference-count":18,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031971952","type":"print"},{"value":"9783031971969","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-031-97196-9_13","type":"book-chapter","created":{"date-parts":[[2025,9,30]],"date-time":"2025-09-30T22:25:39Z","timestamp":1759271139000},"page":"154-164","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Toward Portable GPU Performance: Julia Recursive Implementation of\u00a0TRMM and\u00a0TRSM"],"prefix":"10.1007","author":[{"given":"Vicki","family":"Carric","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Maxwell","family":"Onyango","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rabab","family":"Alomairy","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Evelyne","family":"Ringoot","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"James","family":"Schloss","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alan","family":"Edelman","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,10,1]]},"reference":[{"key":"13_CR1","unstructured":"Alomairy, R., Gates, M., Cayrols, S., Sukkari, D.: Communication avoiding $$\\{$$LU$$\\}$$ with tournament pivoting in $$\\{$$SLATE$$\\}$$,$$\\{$$SWAN$$\\}$$ No. 18. (2022)"},{"key":"13_CR2","doi-asserted-by":"publisher","unstructured":"Alomairy, R., Ringoot, E., Xuan, S., Carrica, V., Onyango, M., Samaroo, J.: NextLA.jl: next-gen linear algebra. Zenodo, https:\/\/doi.org\/10.5281\/zenodo.15049222 (2025)","DOI":"10.5281\/zenodo.15049222"},{"key":"13_CR3","doi-asserted-by":"crossref","unstructured":"Alomairy, R., Tome, F., Samaroo, J., Edelman, A.: Dynamic task scheduling with data dependency awareness using Julia. In: 2024 IEEE High Performance Extreme Computing Conference (HPEC), IEEE (2024)","DOI":"10.1109\/HPEC62836.2024.10938467"},{"key":"13_CR4","unstructured":"Alomairy, R.M.: High-performance scientific applications using mixed precision and low-rank approximation powered by task-based runtime systems (2022)"},{"key":"13_CR5","unstructured":"Besard, T.: oneAPI.jl, January 2025"},{"issue":"4","key":"13_CR6","doi-asserted-by":"publisher","first-page":"827","DOI":"10.1109\/TPDS.2018.2872064","volume":"30","author":"T Besard","year":"2019","unstructured":"Besard, T., Foket, C., De Sutter, B.: Effective extensible programming: unleashing Julia on GPUs. IEEE Trans. Parallel Distrib. Syst. 30(4), 827\u2013841 (2019)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"13_CR7","unstructured":"Besard, T., Hawkins, M.: Metal.jl, January 2025"},{"key":"13_CR8","doi-asserted-by":"crossref","unstructured":"Blackford, L.S., et al.: An Updated set of basic linear algebra subprograms (BLAS). ACM Trans. Math. Softw. 28(2), 135\u2013151 (2002)","DOI":"10.1145\/567806.567807"},{"issue":"22","key":"13_CR9","doi-asserted-by":"publisher","first-page":"e4187","DOI":"10.1002\/cpe.4187","volume":"29","author":"A Charara","year":"2017","unstructured":"Charara, A., Keyes, D., Ltaief, H.: A framework for dense triangular matrix kernels on various manycore architectures. Concurrency Comput. Pract. Experience 29(22), e4187 (2017)","journal-title":"Concurrency Comput. Pract. Experience"},{"key":"13_CR10","doi-asserted-by":"publisher","unstructured":"Charara, A., Ltaief, H., Keyes, D.: Redesigning triangular dense matrix computations on GPUs. In: Dutot, P.-F., Trystram, D. (eds.) Euro-Par 2016. LNCS, vol. 9833, pp. 477\u2013489. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-43659-3_35","DOI":"10.1007\/978-3-319-43659-3_35"},{"key":"13_CR11","unstructured":"Churavy, V.: Language evolution for parallel and scientific computing. Ph.d. thesis, Massachusetts Institute of Technology, Department of Electrical Engineering and Computer Science, Cambridge, MA, September 2024. Licensed under a CC BY-NC-ND 4.0 license"},{"key":"13_CR12","unstructured":"Churavy, V., et al.: Bridging HPC communities through the julia programming language. arXiv preprintarXiv:2211.02740 (2022)"},{"key":"13_CR13","unstructured":"Davis, J.H., et al.: An evaluative comparison of performance portability across GPU programming models. arXiv preprintarXiv:2402.08950 (2024)"},{"issue":"25","key":"13_CR14","doi-asserted-by":"publisher","first-page":"e7811","DOI":"10.1002\/cpe.7811","volume":"35","author":"M Faverge","year":"2023","unstructured":"Faverge, M., et al.: Programming heterogeneous architectures Uing hierarchical tasks. Concurrency Comput. Pract. Experience 35(25), e7811 (2023)","journal-title":"Concurrency Comput. Pract. Experience"},{"key":"13_CR15","doi-asserted-by":"crossref","unstructured":"Gates, M., et al.: Evolution of the SLATE linear algebra library. Int. J. High Perform. Comput. Appl. 39(1), 3\u201317 (2025)","DOI":"10.1177\/10943420241286531"},{"key":"13_CR16","doi-asserted-by":"crossref","unstructured":"Hoque, R., Herault, T., Bosilca, G., Dongarra, J.: Dynamic task discovery in PaRSEC: a data-flow task-based runtime. In: Proceedings of the 8th Workshop on Latest Advances in Scalable Algorithms for Large-Scale Systems, pp. 1\u20138 (2017)","DOI":"10.1145\/3148226.3148233"},{"key":"13_CR17","unstructured":"Samaroo, J., et al.: Juliagpu\/amdgpu.jl: v0.7.3, October 2023"},{"key":"13_CR18","unstructured":"Xuan, S., Ringoot, E., Alomairy, R., Tome, F., Samaroo, J., Edelman, A.: Synthesizing Numerical linear algebra using julia. In: 2024 IEEE High Performance Extreme Computing Conference (HPEC), IEEE (2024)"}],"container-title":["Lecture Notes in Computer Science","Asynchronous Many-Task Systems and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-97196-9_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T03:38:55Z","timestamp":1781667535000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-97196-9_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,1]]},"ISBN":["9783031971952","9783031971969"],"references-count":18,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-97196-9_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,1]]},"assertion":[{"value":"1 October 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"WAMTA","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Workshop on Asynchronous Many-Task Systems and Applications","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"St. Louis, MO","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"USA","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 February 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 February 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"wamta2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/wamta25.github.io\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}