{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T01:45:08Z","timestamp":1787017508091,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["NRF-2022R1C1C1011021, RS-2024-00357037, RS-2025-00553645"],"award-info":[{"award-number":["NRF-2022R1C1C1011021, RS-2024-00357037, RS-2025-00553645"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Microsoft Research PhD Fellowship"},{"name":"Swiss National Science Foundation","award":["200021_212757"],"award-info":[{"award-number":["200021_212757"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,21]]},"DOI":"10.1145\/3695053.3731100","type":"proceedings-article","created":{"date-parts":[[2025,6,20]],"date-time":"2025-06-20T12:46:17Z","timestamp":1750423577000},"page":"153-165","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Avant-Garde: Empowering GPUs with Scaled Numeric Formats"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-8664-9840","authenticated-orcid":false,"given":"Minseong","family":"Gil","sequence":"first","affiliation":[{"name":"Korea University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4090-4025","authenticated-orcid":false,"given":"Dongho","family":"Ha","sequence":"additional","affiliation":[{"name":"MangoBoost Inc., Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1233-5142","authenticated-orcid":false,"given":"Simla Burcu","family":"Harma","sequence":"additional","affiliation":[{"name":"EcoCloud, EPFL, Lausanne, Vaud, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9332-0251","authenticated-orcid":false,"given":"Myung Kuk","family":"Yoon","sequence":"additional","affiliation":[{"name":"Ewha Womans University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5916-8068","authenticated-orcid":false,"given":"Babak","family":"Falsafi","sequence":"additional","affiliation":[{"name":"EcoCloud, EPFL, Lausanne, Vaud, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5390-6445","authenticated-orcid":false,"given":"Won Woo","family":"Ro","sequence":"additional","affiliation":[{"name":"Yonsei University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6442-3705","authenticated-orcid":false,"given":"Yunho","family":"Oh","sequence":"additional","affiliation":[{"name":"Korea University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,20]]},"reference":[{"key":"e_1_3_3_1_2_2","volume-title":"Proceedings of the 33rd International Conference on Neural Information Processing Systems (NIPS)","author":"Banner Ron","year":"2019","unstructured":"Ron Banner, Yury Nahshan, and Daniel Soudry. 2019. Post Training 4-Bit Quantization of Convolutional Networks for Rapid-Deployment. In Proceedings of the 33rd International Conference on Neural Information Processing Systems (NIPS). Curran Associates Inc., Red Hook, NY, USA, Article 714, 9\u00a0pages."},{"key":"e_1_3_3_1_3_2","unstructured":"Tom Savell Ankit More Kyung-Nam Han Ritchie Zhao Mathew Hall Jasmine Klar Eric Chung Yuan Yu Michael Schulte Ralph Wittig Ian Bratt Nigel Stephens Jelena Milanovic John Brothers Pradeep Dubey Marius Cornea Alexander Heinecke Andres Rodriguez Martin Langhammer Summer Deng Maxim Naumov Paulius Micikevicius Michael\u00a0Siu Bita Darvish\u00a0Rouhani Nitin\u00a0Garegrat and Colin Verrilli. 2023. OCP Microscaling Formats (MX) Specification. https:\/\/www.opencompute.org\/documents\/ocp-microscaling-formats-mx-v1-0-spec-final-pdf."},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1145\/3342195.3387555"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589351"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_3_1_7_2","unstructured":"Jacob Devlin Ming-Wei Chang Kenton Lee and Kristina Toutanova. 2018. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. CoRR abs\/1810.04805 (2018). arXiv:https:\/\/arXiv.org\/abs\/1810.04805http:\/\/arxiv.org\/abs\/1810.04805"},{"key":"e_1_3_3_1_8_2","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly Jakob Uszkoreit and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. arxiv:https:\/\/arXiv.org\/abs\/2010.11929\u00a0[cs.CV]"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/3466752.3480057"},{"key":"e_1_3_3_1_10_2","volume-title":"Advances in Neural Information Processing Systems (NIPS)","author":"Drumond Mario","year":"2018","unstructured":"Mario Drumond, Tao LIN, Martin Jaggi, and Babak Falsafi. 2018. Training DNNs with Hybrid Block Floating Point. In Advances in Neural Information Processing Systems (NIPS), Vol.\u00a031. Curran Associates, Inc."},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"crossref","unstructured":"Luciano Floridi and Massimo Chiriatti. 2020. GPT-3: Its nature scope limits and consequences. Minds and Machines 30 4 (2020) 681\u2013694.","DOI":"10.1007\/s11023-020-09548-1"},{"key":"e_1_3_3_1_12_2","unstructured":"Wikimedia Foundation. [n. d.]. Toggle the table of contents English Wikipedia. https:\/\/en.wikipedia.org\/wiki\/English_Wikipedia"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00630"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"publisher","DOI":"10.1145\/3007787.3001163"},{"key":"e_1_3_3_1_15_2","unstructured":"Simla\u00a0Burcu Harma Ayan Chakraborty Nicholas Sperry Babak Falsafi Martin Jaggi and Yunho Oh. 2022. Accuracy Booster: Enabling 4-bit Fixed-point Arithmetic for DNN Training. arxiv:https:\/\/arXiv.org\/abs\/2211.10737\u00a0[cs.LG]"},{"key":"e_1_3_3_1_16_2","unstructured":"Intel. [n. d.]. Intel neural compressor. https:\/\/intel.github.io\/neural-compressor. Accessed: 2025-02-16."},{"key":"e_1_3_3_1_17_2","unstructured":"Intel. [n. d.]. OpenVINO: Open-source software toolkit for optimizing and deploying deep learning models. https:\/\/github.com\/openvinotoolkit\/openvino. Accessed: 2025-02-16."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00286"},{"key":"e_1_3_3_1_19_2","unstructured":"Wonsuk Jang and Thierry Tambe. 2025. BlockDialect: Block-wise Fine-grained Mixed Format Quantization for Energy-Efficient LLM Inference. arxiv:https:\/\/arXiv.org\/abs\/2501.01144\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2501.01144"},{"key":"e_1_3_3_1_20_2","first-page":"132","volume-title":"Proceedings of Machine Learning and Systems (MLSys)","volume":"1","author":"Jayarajan Anand","year":"2019","unstructured":"Anand Jayarajan, Jinliang Wei, Garth Gibson, Alexandra Fedorova, and Gennady Pekhimenko. 2019. Priority-based Parameter Propagation for Distributed DNN Training. In Proceedings of Machine Learning and Systems (MLSys), Vol.\u00a01. 132\u2013145."},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1145\/3673038.3673045"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA45697.2020.00047"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO50266.2020.00065"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2018.00041"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3649329.3657323"},{"key":"e_1_3_3_1_26_2","volume-title":"Advances in Neural Information Processing Systems (NIPS)","author":"K\u00f6ster Urs","year":"2017","unstructured":"Urs K\u00f6ster, Tristan Webb, Xin Wang, Marcel Nassar, Arjun\u00a0K Bansal, William Constable, Oguz Elibol, Scott Gray, Stewart Hall, Luke Hornof, Amir Khosrowshahi, Carey Kloss, Ruby\u00a0J Pai, and Naveen Rao. 2017. Flexpoint: An Adaptive Numerical Format for Efficient Training of Deep Neural Networks. In Advances in Neural Information Processing Systems (NIPS), Vol.\u00a030. Curran Associates, Inc."},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.1109\/DAC56929.2023.10248013"},{"key":"e_1_3_3_1_28_2","first-page":"161","volume-title":"2021 USENIX Annual Technical Conference (USENIX ATC 21)","author":"Lim Gangmuk","year":"2021","unstructured":"Gangmuk Lim, Jeongseob Ahn, Wencong Xiao, Youngjin Kwon, and Myeongjae Jeon. 2021. Zico: Efficient GPU Memory Sharing for Concurrent DNN Training. In 2021 USENIX Annual Technical Conference (USENIX ATC 21). USENIX Association, 161\u2013175. https:\/\/www.usenix.org\/conference\/atc21\/presentation\/lim"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"crossref","unstructured":"Chang Liu Xishan Zhang Rui Zhang Ling Li Shiyi Zhou Di Huang Zhen Li Zidong Du Shaoli Liu and Tianshi Chen. 2022. Rethinking the Importance of Quantization Bias Toward Full Low-Bit Training. IEEE Transactions on Image Processing 31 (2022) 7006\u20137019.","DOI":"10.1109\/TIP.2022.3216776"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614249"},{"key":"e_1_3_3_1_31_2","unstructured":"Paulius Micikevicius Dusan Stosic Neil Burgess Marius Cornea Pradeep Dubey Richard Grisenthwaite Sangwon Ha Alexander Heinecke Patrick Judd John Kamalu Naveen Mellempudi Stuart Oberman Mohammad Shoeybi Michael Siu and Hao Wu. 2022. FP8 Formats for Deep Learning. arxiv:https:\/\/arXiv.org\/abs\/2209.05433\u00a0[cs.LG]"},{"key":"e_1_3_3_1_32_2","unstructured":"Microsoft. 2024. MicroScaling Emulator. https:\/\/github.com\/microsoft\/microxcaling."},{"key":"e_1_3_3_1_33_2","series-title":"Proceedings of Machine Learning Research","first-page":"7937","volume-title":"Proceedings of the 38th International Conference on Machine Learning","volume":"139","author":"Narayanan Deepak","year":"2021","unstructured":"Deepak Narayanan, Amar Phanishayee, Kaiyu Shi, Xie Chen, and Matei Zaharia. 2021. Memory-Efficient Pipeline-Parallel DNN Training. In Proceedings of the 38th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a0139). PMLR, 7937\u20137947."},{"key":"e_1_3_3_1_34_2","unstructured":"NVIDIA. 2020. NVIDIA A100 Tensor Core GPU Architecture. v1.0 (2020) 1\u201382."},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2018.00037"},{"key":"e_1_3_3_1_36_2","unstructured":"OpenAI. 2023. GPT-4 Technical Report. arxiv:https:\/\/arXiv.org\/abs\/2303.08774\u00a0[cs.CL]"},{"key":"e_1_3_3_1_37_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01225-0_36"},{"key":"e_1_3_3_1_38_2","first-page":"307","volume-title":"2020 USENIX Annual Technical Conference (USENIX ATC 20)","author":"Park Jay\u00a0H.","year":"2020","unstructured":"Jay\u00a0H. Park, Gyeongchan Yun, Chang\u00a0M. Yi, Nguyen\u00a0T. Nguyen, Seungmin Lee, Jaesik Choi, Sam\u00a0H. Noh, and Young ri Choi. 2020. HetPipe: Enabling Large DNN Training on (Whimpy) Heterogeneous GPU Clusters through Integration of Pipelined Model Parallelism and Data Parallelism. In 2020 USENIX Annual Technical Conference (USENIX ATC 20). USENIX Association, 307\u2013321. https:\/\/www.usenix.org\/conference\/atc20\/presentation\/park"},{"key":"e_1_3_3_1_39_2","unstructured":"Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei and Ilya Sutskever. 2018. Language Models are Unsupervised Multitask Learners. (2018)."},{"key":"e_1_3_3_1_40_2","unstructured":"Akshat Ramachandran Souvik Kundu and Tushar Krishna. 2024. MicroScopiQ: Accelerating Foundational Models through Outlier-Aware Microscaling Quantization. arxiv:https:\/\/arXiv.org\/abs\/2411.05282\u00a0[cs.AR] https:\/\/arxiv.org\/abs\/2411.05282"},{"key":"e_1_3_3_1_41_2","unstructured":"MIT\u00a0Technology Review. 2023. Multi-die systems define the future of semiconductors. https:\/\/www.technologyreview.com\/2023\/03\/31\/1070527\/multi-die-systems-define-the-future-of-semiconductors\/. Accessed: 2023-03-31."},{"key":"e_1_3_3_1_42_2","volume-title":"NeurIPS 2020","author":"Rouhani Bita","year":"2020","unstructured":"Bita Rouhani, Daniel Lo, Ritchie Zhao, Ming Liu, Jeremy Fowers, Kalin Ovtcharov, Anna Vinogradsky, Sarah Massengill, Lita Yang, Ray Bittner, Alessandro Forin, Haishan Zhu, Taesik Na, Prerak Patel, Shuai Che, Lok\u00a0Chand Koppaka, Xia Song, Subhojit Som, Kaustav Das, Saurabh Tiwary, Steve Reinhardt, Sitaram Lanka, Eric Chung, and Doug Burger. 2020. Pushing the Limits of Narrow Precision Inferencing at Cloud Scale with Microsoft Floating Point. In NeurIPS 2020. ACM."},{"key":"e_1_3_3_1_43_2","unstructured":"Bita\u00a0Darvish Rouhani Ritchie Zhao Ankit More Mathew Hall Alireza Khodamoradi Summer Deng Dhruv Choudhary Marius Cornea Eric Dellinger Kristof Denolf Stosic Dusan Venmugil Elango Maximilian Golub Alexander Heinecke Phil James-Roxby Dharmesh Jani Gaurav Kolhe Martin Langhammer Ada Li Levi Melnick Maral Mesmakhosroshahi Andres Rodriguez Michael Schulte Rasoul Shafipour Lei Shao Michael Siu Pradeep Dubey Paulius Micikevicius Maxim Naumov Colin Verrilli Ralph Wittig Doug Burger and Eric Chung. 2023. Microscaling Data Formats for Deep Learning. arxiv:https:\/\/arXiv.org\/abs\/2310.10537\u00a0[cs.LG]"},{"key":"e_1_3_3_1_44_2","series-title":"Proceedings of Machine Learning Research","first-page":"241","volume-title":"Proceedings of The 4th NeurIPS Efficient Natural Language and Speech Processing Workshop","volume":"262","author":"Sharify Sayeh","year":"2024","unstructured":"Sayeh Sharify, Utkarsh Saxena, Zifei Xu, Wanzin Yazar, Ilya Soloveychik, and Xin Wang. 2024. Post Training Quantization of Large Language Models with Microscaling Formats. In Proceedings of The 4th NeurIPS Efficient Natural Language and Speech Processing Workshop(Proceedings of Machine Learning Research, Vol.\u00a0262), Mehdi Rezagholizadeh, Peyman Passban, Soheila Samiee, Vahid Partovi\u00a0Nia, Yu\u00a0Cheng, Yue Deng, Qun Liu, and Boxing Chen (Eds.). PMLR, 241\u2013258. https:\/\/proceedings.mlr.press\/v262\/sharify24a.html"},{"key":"e_1_3_3_1_45_2","doi-asserted-by":"publisher","DOI":"10.1109\/MSE.2007.44"},{"key":"e_1_3_3_1_46_2","first-page":"1796","volume-title":"Advances in Neural Information Processing Systems (NIPS)","author":"Sun Xiao","year":"2020","unstructured":"Xiao Sun, Naigang Wang, Chia-Yu Chen, Jiamin Ni, Ankur Agrawal, Xiaodong Cui, Swagath Venkataramani, Kaoutar El\u00a0Maghraoui, Vijayalakshmi\u00a0(Viji) Srinivasan, and Kailash Gopalakrishnan. 2020. Ultra-Low Precision 4-bit Training of Deep Neural Networks. In Advances in Neural Information Processing Systems (NIPS), Vol.\u00a033. Curran Associates, Inc., 1796\u20131807."},{"key":"e_1_3_3_1_47_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614247"},{"key":"e_1_3_3_1_48_2","doi-asserted-by":"publisher","DOI":"10.1145\/3178487.3178491"},{"key":"e_1_3_3_1_49_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503221.3508408"},{"key":"e_1_3_3_1_50_2","doi-asserted-by":"publisher","DOI":"10.1145\/3492321.3519557"},{"key":"e_1_3_3_1_51_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00067"},{"key":"e_1_3_3_1_52_2","doi-asserted-by":"crossref","unstructured":"Kang Zhao Sida Huang Pan Pan Yinghan Li Yingya Zhang Zhenyu Gu and Yinghui Xu. 2021. Distribution Adaptive INT8 Quantization for Training CNNs. Proceedings of the AAAI Conference on Artificial Intelligence 35 4 (May 2021) 3483\u20133491.","DOI":"10.1609\/aaai.v35i4.16462"},{"key":"e_1_3_3_1_53_2","volume-title":"5th International Conference on Learning Representations, ICLR 2017, Toulon, France, April 24-26, 2017, Conference Track Proceedings","author":"Zhou Aojun","year":"2017","unstructured":"Aojun Zhou, Anbang Yao, Yiwen Guo, Lin Xu, and Yurong Chen. 2017. Incremental Network Quantization: Towards Lossless CNNs with Low-precision Weights. In 5th International Conference on Learning Representations, ICLR 2017, Toulon, France, April 24-26, 2017, Conference Track Proceedings."}],"event":{"name":"ISCA '25: Proceedings of the 52nd Annual International Symposium on Computer Architecture","location":"Tokyo Japan","acronym":"SIGARCH '25","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 52nd Annual International Symposium on Computer Architecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3695053.3731100","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T07:03:27Z","timestamp":1750489407000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3695053.3731100"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,20]]},"references-count":52,"alternative-id":["10.1145\/3695053.3731100","10.1145\/3695053"],"URL":"https:\/\/doi.org\/10.1145\/3695053.3731100","relation":{},"subject":[],"published":{"date-parts":[[2025,6,20]]},"assertion":[{"value":"2025-06-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}