{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T14:30:16Z","timestamp":1787495416185,"version":"build-2736575974"},"publisher-location":"Singapore","reference-count":18,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819248049","type":"print"},{"value":"9789819248056","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-4805-6_47","type":"book-chapter","created":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:48:23Z","timestamp":1787492903000},"page":"632-637","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["COMET: An FP32 Matrix Multiplication Accelerator by\u00a0Extending INT8-Based Arrays"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-4235-0893","authenticated-orcid":false,"given":"Xia","family":"Ruiyang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9823-2573","authenticated-orcid":false,"given":"Hao","family":"Yifan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3529-6319","authenticated-orcid":false,"given":"Zhao","family":"Yuming","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2285-534X","authenticated-orcid":false,"given":"Liu","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5503-4457","authenticated-orcid":false,"given":"Zhao","family":"Yongwei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7603-4210","authenticated-orcid":false,"given":"Du","family":"Zidong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9979-0561","authenticated-orcid":false,"given":"Hu","family":"Xing","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-0237-1034","authenticated-orcid":false,"given":"Li","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2530-5874","authenticated-orcid":false,"given":"Guo","family":"Qi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,24]]},"reference":[{"key":"47_CR1","first-page":"223","volume":"3","author":"H Abdelaziz","year":"2021","unstructured":"Abdelaziz, H., Shin, J.H., Pedram, A., Hassoun, J., et al.: Rethinking floating point overheads for mixed precision DNN accelerators. Proc. Mach. Learn. Syst. 3, 223\u2013239 (2021)","journal-title":"Proc. Mach. Learn. Syst."},{"key":"47_CR2","doi-asserted-by":"publisher","first-page":"291","DOI":"10.1090\/S0002-9947-1969-0249212-8","volume":"142","author":"SA Cook","year":"1969","unstructured":"Cook, S.A., Aanderaa, S.O.: On the minimum computation time of functions. Trans. Am. Math. Soc. 142, 291\u2013314 (1969)","journal-title":"Trans. Am. Math. Soc."},{"issue":"1","key":"47_CR3","first-page":"1","volume":"19","author":"Y Fu","year":"2021","unstructured":"Fu, Y., Bolotin, E., Chatterjee, N., Nellans, D., Keckler, S.W.: GPU domain specialization via composable on-package architecture. ACM Trans. Architecture Code Optimization (TACO) 19(1), 1\u201323 (2021)","journal-title":"ACM Trans. Architecture Code Optimization (TACO)"},{"key":"47_CR4","doi-asserted-by":"crossref","unstructured":"Ha, D., Zhang, Y., Kao, C.C., Hughes, C.J., Ro, W.W., Tseng, H.W.: M 3 xu: achieving high-precision and complex matrix multiplication with low-precision MXUS. In: SC24: International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201316. IEEE (2024)","DOI":"10.1109\/SC41406.2024.00016"},{"key":"47_CR5","unstructured":"Jouppi, N.P., et al.: In-datacenter performance analysis of a tensor processing unit. In: Proceedings of the 44th Annual International Symposium on Computer Architecture, pp. 1\u201312 (2017)"},{"key":"47_CR6","doi-asserted-by":"crossref","unstructured":"Kulisch, U.W., Miranker, W.L.: The arithmetic of the digital computer: a new approach. SIAM Rev. 28(1), 1\u201340 (1986)","DOI":"10.1137\/1028001"},{"key":"47_CR7","doi-asserted-by":"crossref","unstructured":"Lin, Z., Sun, A., Zhang, X., Lu, Y.: Mixpert: Optimizing mixed-precision floating point emulation on GPU integer tensor cores. In: Proceedings of the 25th ACM SIGPLAN\/SIGBED International Conference on Languages, Compilers, and Tools for Embedded Systems, pp. 34\u201345 (2024)","DOI":"10.1145\/3652032.3657567"},{"key":"47_CR8","doi-asserted-by":"crossref","unstructured":"Ma, Z., et al.: Efficiently emulating high-bitwidth computation with low-bitwidth hardware. In: Proceedings of the 36th ACM International Conference on Supercomputing, pp. 1\u201312 (2022)","DOI":"10.1145\/3524059.3532377"},{"key":"47_CR9","unstructured":"Mudigere, D., et al.: Software-hardware co-design for fast and scalable training of deep learning recommendation models. In: Proceedings of the 49th Annual International Symposium on Computer Architecture, pp. 993\u20131011 (2022)"},{"issue":"9","key":"47_CR10","doi-asserted-by":"publisher","first-page":"2522","DOI":"10.1109\/TC.2023.3253050","volume":"72","author":"SH Noh","year":"2023","unstructured":"Noh, S.H., Koo, J., Lee, S., Park, J., Kung, J.: Flexblock: a flexible DNN training accelerator with multi-mode block floating point support. IEEE Trans. Comput. 72(9), 2522\u20132535 (2023)","journal-title":"IEEE Trans. Comput."},{"key":"47_CR11","unstructured":"NVIDIA Corporation: NVIDIA A100 Tensor Core GPU Architecture: Unprecedented Acceleration at Every Scale. Whitepaper, NVIDIA Corporation, US (2020)"},{"key":"47_CR12","unstructured":"NVIDIA Corporation: NVIDIA H100 Tensor Core GPU Architecture: Exceptional Performance, Scalability, and Security for The Data Center. Whitepaper, NVIDIA Corporation, US (2022)"},{"key":"47_CR13","first-page":"714","volume":"3","author":"AL Toom","year":"1963","unstructured":"Toom, A.L.: The complexity of a scheme of functional elements realizing the multiplication of integers. Soviet Math. Doklady 3, 714\u2013716 (1963)","journal-title":"Soviet Math. Doklady"},{"key":"47_CR14","doi-asserted-by":"crossref","unstructured":"Valero-Lara, P., Jorquera, I., Lui, F., Vetter, J.: Mixed-precision s\/DGEMM using the tf32 and tf64 frameworks on low-precision ai tensor cores. In: Proceedings of the SC\u201923 Workshops of the International Conference on High Performance Computing, Network, Storage, and Analysis, pp. 179\u2013186 (2023)","DOI":"10.1145\/3624062.3624084"},{"key":"47_CR15","doi-asserted-by":"publisher","first-page":"178","DOI":"10.1016\/j.cpc.2018.03.016","volume":"228","author":"H Wang","year":"2018","unstructured":"Wang, H., Zhang, L., Han, J., et al.: Deepmd-kit: a deep learning package for many-body potential energy representation and molecular dynamics. Comput. Phys. Commun. 228, 178\u2013184 (2018)","journal-title":"Comput. Phys. Commun."},{"key":"47_CR16","unstructured":"Yang, A., et al.: Qwen2.5 Technical report. arXiv preprint arXiv:2412.15115 (2024)"},{"key":"47_CR17","doi-asserted-by":"crossref","unstructured":"Zhang, L., Han, J., Wang, H., Car, R., E, W.: Deep potential molecular dynamics: a scalable model with the accuracy of quantum mechanics. Phy. Rev. Lett. 120(14), 143001 (2018)","DOI":"10.1103\/PhysRevLett.120.143001"},{"key":"47_CR18","doi-asserted-by":"crossref","unstructured":"Zhang, S.Q., McDanel, B., Kung, H.: Fast: DNN training under variable precision block floating point with stochastic rounding. In: 2022 IEEE International Symposium on High-Performance Computer Architecture (HPCA), pp. 846\u2013860. IEEE (2022)","DOI":"10.1109\/HPCA53966.2022.00067"}],"container-title":["Lecture Notes in Computer Science","Advanced Parallel Processing Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-4805-6_47","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:48:26Z","timestamp":1787492906000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-4805-6_47"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,24]]},"ISBN":["9789819248049","9789819248056"],"references-count":18,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-4805-6_47","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,24]]},"assertion":[{"value":"24 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","label":"Disclosure of Interests","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"APPT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Advanced Parallel Processing Technologies","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brussels","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Belgium","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"appt2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.appt-conference.com\/2026","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}