{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T14:30:10Z","timestamp":1787495410954,"version":"build-2736575974"},"publisher-location":"Singapore","reference-count":17,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819248049","type":"print"},{"value":"9789819248056","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-4805-6_22","type":"book-chapter","created":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:48:33Z","timestamp":1787492913000},"page":"330-342","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Fold-FP4: Folding for\u00a0Energy Efficient FP4 Matrix Multiplication"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-3645-5708","authenticated-orcid":false,"given":"Qiyan","family":"Fang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4017-8913","authenticated-orcid":false,"given":"Yifan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5503-4457","authenticated-orcid":false,"given":"Yongwei","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yi","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-3128-4618","authenticated-orcid":false,"given":"Zetao","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,24]]},"reference":[{"key":"22_CR1","unstructured":"AMD: AMD Instinct MI350 Series GPUs. AMD Official Product Page (2025)"},{"key":"22_CR2","doi-asserted-by":"publisher","unstructured":"Balasubramonian, R., Kahng, A.B., Muralimanohar, N., Shafiee, A., Srinivas, V.: CACTI 7: new tools for interconnect exploration in innovative off-chip memories. ACM Trans. Archit. Code Optim. 14(2), 14:1\u201314:25 (2017). https:\/\/doi.org\/10.1145\/3085572","DOI":"10.1145\/3085572"},{"key":"22_CR3","doi-asserted-by":"publisher","unstructured":"Chen, T., Du, Z., Sun, N., Wang, J., Wu, C., Chen, Y., et\u00a0al.: DianNao: a small-footprint high-throughput accelerator for ubiquitous machine-learning. In: Proceedings of the 19th International Conference on Architectural Support for Programming Languages and Operating Systems, pp. 269\u2013284 (2014). https:\/\/doi.org\/10.1145\/2541940.2541967","DOI":"10.1145\/2541940.2541967"},{"key":"22_CR4","doi-asserted-by":"publisher","unstructured":"Chen, Y., Zhao, Y., Hao, Y., Wen, Y., Dai, Y., Li, X., et\u00a0al.: Cambricon-c: efficient 4-bit matrix unit via primitivization. In: Proceedings of the 2024 57th IEEE\/ACM International Symposium on Microarchitecture, pp. 538\u2013550 (2024). https:\/\/doi.org\/10.1109\/MICRO61859.2024.00047","DOI":"10.1109\/MICRO61859.2024.00047"},{"key":"22_CR5","doi-asserted-by":"crossref","unstructured":"Cuyckens, S., Yi, X., Murthy, N.S., Fang, C., Verhelst, M.: Efficient precision-scalable hardware for microscaling (MX) processing in robotics learning. arXiv preprint arXiv:2505.22404 (2025)","DOI":"10.1109\/ISLPED65674.2025.11261796"},{"key":"22_CR6","doi-asserted-by":"publisher","unstructured":"Guo, C., Zhang, C., Leng, J., Liu, Z., Yang, F., Liu, Y., et\u00a0al.: ANT: exploiting adaptive numerical data type for low-bit deep neural network quantization. In: Proceedings of the 55th IEEE\/ACM International Symposium on Microarchitecture, pp. 1414\u20131433 (2022). https:\/\/doi.org\/10.1109\/MICRO56248.2022.00097","DOI":"10.1109\/MICRO56248.2022.00097"},{"key":"22_CR7","doi-asserted-by":"publisher","unstructured":"Hegde, K., Yu, J., Agrawal, R., Yan, M., Pellauer, M., Fletcher, C.W.: UCNN: exploiting computational reuse in deep neural networks via weight repetition. In: Proceedings of the 2018 ACM\/IEEE 45th Annual International Symposium on Computer Architecture, pp. 674\u2013687 (2018). https:\/\/doi.org\/10.1109\/ISCA.2018.00062","DOI":"10.1109\/ISCA.2018.00062"},{"key":"22_CR8","unstructured":"Hoffmann, J.: Training compute-optimal large language models. arXiv preprint arXiv:2203.15556 (2022)"},{"key":"22_CR9","doi-asserted-by":"publisher","unstructured":"Hu, W., et al.: M2XFP: a metadata-augmented microscaling data format for efficient low-bit quantization. In: Proceedings of the 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems (2026). https:\/\/doi.org\/10.1145\/3779212.3790185","DOI":"10.1145\/3779212.3790185"},{"key":"22_CR10","doi-asserted-by":"publisher","unstructured":"Judd, P., Albericio, J., Hetherington, T.H., Aamodt, T.M., Moshovos, A.: Stripes: Bit-serial deep neural network computing. In: Proceedings of the 2016 49th Annual IEEE\/ACM International Symposium on Microarchitecture, pp. 1\u201312 (2016). https:\/\/doi.org\/10.1109\/MICRO.2016.7783722","DOI":"10.1109\/MICRO.2016.7783722"},{"key":"22_CR11","doi-asserted-by":"publisher","unstructured":"Kumar, S., Sharma, V.P., Neema, V., Vishvakarma, S.K.: DHFP-PE: dual-precision hybrid floating point processing element for ai acceleration. arXiv preprint arXiv:2604.04507 (2026). https:\/\/doi.org\/10.48550\/arXiv.2604.04507","DOI":"10.48550\/arXiv.2604.04507"},{"key":"22_CR12","unstructured":"Lin, J., Tang, J., Tang, H., Yang, S., Chen, W.M., Wang, W.C., et\u00a0al.: AWQ: activation-aware weight quantization for LLM compression and acceleration. In: Proceedings of Machine Learning and Systems (2024)"},{"key":"22_CR13","doi-asserted-by":"crossref","unstructured":"Liu, S.y., Liu, Z., Huang, X., Dong, P., Cheng, K.T.: LLM-FP4: 4-bit floating-point quantized transformers. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.39"},{"key":"22_CR14","doi-asserted-by":"publisher","unstructured":"Mo, Z., Wang, L., Wei, J., Zeng, Z., Cao, S., Ma, L., et\u00a0al.: LUT tensor core: a software-hardware co-design for LUT-based low-bit LLM inference. In: Proceedings of the 52nd Annual International Symposium on Computer Architecture, pp. 514\u2013528 (2025). https:\/\/doi.org\/10.1145\/3695053.3731057","DOI":"10.1145\/3695053.3731057"},{"key":"22_CR15","unstructured":"NVIDIA: NVIDIA TensorRT unlocks fp4 image generation for NVIDIA blackwell geforce RTX 50 series GPUs. NVIDIA Technical Blog (2025)"},{"key":"22_CR16","doi-asserted-by":"publisher","unstructured":"Pan, Z., San\u00a0Miguel, J., Wu, D.: Carat: unlocking value-level parallelism for multiplier-free GEMMs. In: Proceedings of the 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, vol. 2, pp. 167\u2013184 (2024). https:\/\/doi.org\/10.1145\/3620665.3640364","DOI":"10.1145\/3620665.3640364"},{"key":"22_CR17","unstructured":"Xiao, G., Lin, J., Seznec, M., Wu, H., Demouth, J., Han, S.: SmoothQuant: accurate and efficient post-training quantization for large language models. In: Proceedings of the 40th International Conference on Machine Learning (2023)"}],"container-title":["Lecture Notes in Computer Science","Advanced Parallel Processing Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-4805-6_22","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:48:35Z","timestamp":1787492915000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-4805-6_22"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,24]]},"ISBN":["9789819248049","9789819248056"],"references-count":17,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-4805-6_22","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,24]]},"assertion":[{"value":"24 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","label":"Disclosure of Interests","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"Artificial intelligence tools, if used, were used only for language organization and figure\/table polishing, and did not participate in the generation of the core ideas or technical content.","order":2,"name":"Ethics","label":"Use of Artificial Intelligence Tools","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"APPT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Advanced Parallel Processing Technologies","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brussels","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Belgium","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"appt2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.appt-conference.com\/2026","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}