{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T08:00:14Z","timestamp":1776931214889,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":74,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T00:00:00Z","timestamp":1760659200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"DoE Office of Science","award":["DE-SC0024079"],"award-info":[{"award-number":["DE-SC0024079"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,18]]},"DOI":"10.1145\/3725843.3756118","type":"proceedings-article","created":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T17:19:56Z","timestamp":1760721596000},"page":"869-883","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["MX+: Pushing the Limits of Microscaling Formats for Efficient Large Language Model Serving"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-5629-6258","authenticated-orcid":false,"given":"Jungi","family":"Lee","sequence":"first","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-4228-1138","authenticated-orcid":false,"given":"Junyong","family":"Park","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8624-9913","authenticated-orcid":false,"given":"Soohyun","family":"Cha","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-6853-368X","authenticated-orcid":false,"given":"Jaehoon","family":"Cho","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0403-9928","authenticated-orcid":false,"given":"Jaewoong","family":"Sim","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,17]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"Marah Abdin Jyoti Aneja Harkirat Behl S\u00e9\u00a0bastien Bubeck Ronen Eldan Suriya Gunasekar Michael Harrison Russell\u00a0J Hewett Mojan Javaheripi Piero Kauffmann et\u00a0al. 2024. Phi-4 Technical Report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.08905 (2024)."},{"key":"e_1_3_3_2_3_2","unstructured":"AMD. 2025. AMD Instinct MI355X GPUs. https:\/\/www.amd.com\/en\/products\/accelerators\/instinct\/mi350\/mi355x.html"},{"key":"e_1_3_3_2_4_2","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Ashkboos Saleh","year":"2024","unstructured":"Saleh Ashkboos, Amirkeivan Mohtashami, Maximilian\u00a0L. Croci, Bo Li, Pashmina Cameron, Martin Jaggi, Dan Alistarh, Torsten Hoefler, and James Hensman. 2024. QuaRot: Outlier-Free 4-Bit Inference in Rotated LLMs. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","DOI":"10.1145\/3007787.3001177"},{"key":"e_1_3_3_2_6_2","volume-title":"Proceedings of Machine Learning and Systems (MLSys)","author":"Dai Steve","year":"2021","unstructured":"Steve Dai, Rangha Venkatesan, Mark Ren, Brian Zimmer, William Dally, and Brucek Khailany. 2021. VS-Quant: Per-vector Scaled Quantization for Accurate Low-Precision Neural Network Inference. In Proceedings of Machine Learning and Systems (MLSys)."},{"key":"e_1_3_3_2_7_2","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Darvish\u00a0Rouhani Bita","year":"2020","unstructured":"Bita Darvish\u00a0Rouhani, Daniel Lo, Ritchie Zhao, Ming Liu, Jeremy Fowers, Kalin Ovtcharov, Anna Vinogradsky, Sarah Massengill, Lita Yang, Ray Bittner, Alessandro Forin, Haishan Zhu, Taesik Na, Prerak Patel, Shuai Che, Lok\u00a0Chand Koppaka, Xia Song, Subhojit Som, Kaustav Das, Saurabh Tiwary, Steve Reinhardt, Sitaram Lanka, Eric Chung, and Doug Burger. 2020. Pushing the Limits of Narrow Precision Inferencing at Cloud Scale with Microsoft Floating Point. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589351"},{"key":"e_1_3_3_2_9_2","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Dettmers Tim","year":"2023","unstructured":"Tim Dettmers, Artidoro Pagnoni, Ari Holtzman, and Luke Zettlemoyer. 2023. QLoRA: Efficient Finetuning of Quantized LLMs. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/3123939.3124552"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.98"},{"key":"e_1_3_3_2_12_2","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Dong Peiyan","year":"2023","unstructured":"Peiyan Dong, Lei Lu, Chao Wu, Cheng Lyu, Geng Yuan, Hao Tang, and Yanzhi Wang. 2023. Packqvit: Faster Sub-8-bit Vision Transformers via Full and Packed Quantization on the Mobile. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.5555\/3326943.3326985"},{"key":"e_1_3_3_2_14_2","unstructured":"Abhimanyu Dubey et\u00a0al. 2024. The Llama 3 Herd of Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.21783 (2024)."},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2018.00012"},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589038"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO56248.2022.00095"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"publisher","DOI":"10.1145\/3007787.3001163"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3623775"},{"key":"e_1_3_3_2_21_2","unstructured":"Intel. 2023. Floating-Point Reference Sheet for Intel\u00ae Architecture. https:\/\/www.intel.com\/content\/www\/us\/en\/content-details\/786447\/floating-point-reference-sheet-for-intel-architecture.html"},{"key":"e_1_3_3_2_22_2","volume-title":"Intel Unleashes Enterprise AI with Gaudi 3, AI Open Systems Strategy and New Customer Wins","year":"2024","unstructured":"Intel. 2024. Intel Unleashes Enterprise AI with Gaudi 3, AI Open Systems Strategy and New Customer Wins. https:\/\/www.intc.com\/news-events\/press-releases\/detail\/1689\/intel-unleashes-enterprise-ai-with-gaudi-3-ai-open-systems"},{"key":"e_1_3_3_2_23_2","unstructured":"Zhe Jia Marco Maggioni Jeffrey Smith and Daniele\u00a0Paolo Scarpazza. 2019. Dissecting the NVidia Turing T4 GPU via Microbenchmarking. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1903.07486 (2019)."},{"key":"e_1_3_3_2_24_2","unstructured":"Albert\u00a0Q. Jiang Alexandre Sablayrolles Arthur Mensch Chris Bamford Devendra\u00a0Singh Chaplot Diego de\u00a0las Casas Florian Bressand Gianna Lengyel Guillaume Lample Lucile Saulnier L\u00e9lio\u00a0Renard Lavaud Marie-Anne Lachaux Pierre Stock Teven\u00a0Le Scao Thibaut Lavril Thomas Wang Timoth\u00e9e Lacroix and William\u00a0El Sayed. 2023. Mistral 7B. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.06825 (2023)."},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589350"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080246"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1109\/HCS61935.2024.10665178"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA45697.2020.00047"},{"key":"e_1_3_3_2_29_2","volume-title":"Proceedings of the International Conference on Machine Learning (ICML)","author":"Kim Sehoon","year":"2024","unstructured":"Sehoon Kim, Coleman Hooper, Amir Gholami, Zhen Dong, Xiuyu Li, Sheng Shen, Michael\u00a0W. Mahoney, and Kurt Keutzer. 2024. SqueezeLLM: Dense-and-Sparse Quantization. In Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00093"},{"key":"e_1_3_3_2_31_2","volume-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems (NIPS)","author":"K\u00f6ster Urs","year":"2017","unstructured":"Urs K\u00f6ster, Tristan\u00a0J. Webb, Xin Wang, Marcel Nassar, Arjun\u00a0K. Bansal, William\u00a0H. Constable, O\u011fuz\u00a0H. Elibol, Scott Gray, Stewart Hall, Luke Hornof, Amir Khosrowshahi, Carey Kloss, Ruby\u00a0J. Pai, and Naveen Rao. 2017. Flexpoint: An Adaptive Numerical Format for Efficient Training of Deep Neural Networks. In Proceedings of the 31st International Conference on Neural Information Processing Systems (NIPS)."},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00080"},{"key":"e_1_3_3_2_34_2","volume-title":"Proceedings of the USENIX Symposium on Operating Systems Design and Implementation (OSDI)","author":"Lee Wonbeom","year":"2024","unstructured":"Wonbeom Lee, Jungi Lee, Junghwan Seo, and Jaewoong Sim. 2024. InfiniGen: Efficient Generative Inference of Large Language Models with Dynamic KV Cache Management. In Proceedings of the USENIX Symposium on Operating Systems Design and Implementation (OSDI)."},{"key":"e_1_3_3_2_35_2","volume-title":"Proceedings of Machine Learning and Systems (MLSys)","author":"Lin Ji","year":"2023","unstructured":"Ji Lin, Jiaming Tang, Haotian Tang, Shang Yang, Xingyu Dang, and Song Han. 2023. AWQ: Activation-aware Weight Quantization for LLM Compression and Acceleration. In Proceedings of Machine Learning and Systems (MLSys)."},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.39"},{"key":"e_1_3_3_2_37_2","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR)","author":"Lo Yun-Chen","year":"2023","unstructured":"Yun-Chen Lo, Tse-Kuang Lee, and Ren-Shuo Liu. 2023. Block and Subword-Scaling Floating-Point (BSFP): An Efficient Non-Uniform Quantization For Low Precision Inference. In Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614249"},{"key":"e_1_3_3_2_39_2","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR)","author":"Merity Stephen","year":"2017","unstructured":"Stephen Merity, Caiming Xiong, James Bradbury, and Richard Socher. 2017. Pointer Sentinel Mixture Models. In Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_2_40_2","unstructured":"Microsoft. 2024. MX PyTorch Emulation Library. https:\/\/github.com\/microsoft\/microxcaling"},{"key":"e_1_3_3_2_41_2","unstructured":"NVIDIA. 2022. NVIDIA RTX A6000 Graphics Card. https:\/\/www.nvidia.com\/en-us\/design-visualization\/rtx-a6000\/"},{"key":"e_1_3_3_2_42_2","unstructured":"NVIDIA. 2024. NVIDIA Blackwell Architecture Technical Brief. https:\/\/resources.nvidia.com\/en-us-blackwell-architecture"},{"key":"e_1_3_3_2_43_2","unstructured":"NVIDIA. 2025. Block Scaling. https:\/\/docs.nvidia.com\/deeplearning\/cudnn\/frontend\/latest\/operations\/BlockScaling.html"},{"key":"e_1_3_3_2_44_2","volume-title":"CUDA Binary Utilities","year":"2025","unstructured":"NVIDIA. 2025. CUDA Binary Utilities. https:\/\/docs.nvidia.com\/cuda\/cuda-binary-utilities\/#nvdisasm"},{"key":"e_1_3_3_2_45_2","volume-title":"NVIDIA CUTLASS Documentation","year":"2025","unstructured":"NVIDIA. 2025. NVIDIA CUTLASS Documentation. https:\/\/docs.nvidia.com\/cutlass\/media\/docs\/cpp\/blackwell_functionality.html#tile-size"},{"key":"e_1_3_3_2_46_2","unstructured":"NVIDIA. 2025. NVIDIA RTX Blackwell GPU Architecture. https:\/\/images.nvidia.com\/aem-dam\/Solutions\/geforce\/blackwell\/nvidia-rtx-blackwell-gpu-architecture.pdf"},{"key":"e_1_3_3_2_47_2","volume-title":"Parallel Thread Execution ISA Version 8.7","year":"2025","unstructured":"NVIDIA. 2025. Parallel Thread Execution ISA Version 8.7. https:\/\/docs.nvidia.com\/cuda\/parallel-thread-execution\/#warp-level-matrix-fragment-mma-16864"},{"key":"e_1_3_3_2_48_2","volume-title":"OCP Microscaling Formats (MX) Specification Version 1.0","year":"2023","unstructured":"Open Compute Project 2023. OCP Microscaling Formats (MX) Specification Version 1.0. Open Compute Project."},{"key":"e_1_3_3_2_49_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2018.00063"},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00015"},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2019.00016"},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"publisher","DOI":"10.1145\/3007787.3001165"},{"key":"e_1_3_3_2_53_2","unstructured":"Bita\u00a0Darvish Rouhani Ritchie Zhao Ankit More Mathew Hall Alireza Khodamoradi Summer Deng Dhruv Choudhary Marius Cornea Eric Dellinger Kristof Denolf Stosic Dusan Venmugil Elango Maximilian Golub Alexander Heinecke Phil James-Roxby Dharmesh Jani Gaurav Kolhe Martin Langhammer Ada Li Levi Melnick Maral Mesmakhosroshahi Andres Rodriguez Michael Schulte Rasoul Shafipour Lei Shao Michael Siu Pradeep Dubey Paulius Micikevicius Maxim Naumov Colin Verrilli Ralph Wittig Doug Burger and Eric Chung. 2023. Microscaling Data Formats for Deep Learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.10537 (2023)."},{"key":"e_1_3_3_2_54_2","doi-asserted-by":"crossref","unstructured":"Olga Russakovsky Jia Deng Hao Su Jonathan Krause Sanjeev Satheesh Sean Ma Zhiheng Huang Andrej Karpathy Aditya Khosla Michael Bernstein et\u00a0al. 2015. Imagenet Large Scale Visual Recognition Challenge. International Journal of Computer Vision (2015).","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA45697.2020.00086"},{"key":"e_1_3_3_2_56_2","doi-asserted-by":"publisher","DOI":"10.1145\/3466752.3480095"},{"key":"e_1_3_3_2_57_2","volume-title":"CUTLASS","author":"Thakkar Vijay","year":"2025","unstructured":"Vijay Thakkar, Pradeep Ramani, Cris Cecka, Aniket Shivam, Honghao Lu, Ethan Yan, Jack Kosaian, Mark Hoemmen, Haicheng Wu, Andrew Kerr, Matt Nicely, Duane Merrill, Dustyn Blasig, Fengqi Qiao, Piotr Majcher, Paul Springer, Markus Hohnerbach, Jin Wang, and Manish Gupta. 2025. CUTLASS. https:\/\/github.com\/NVIDIA\/cutlass"},{"key":"e_1_3_3_2_58_2","doi-asserted-by":"publisher","DOI":"10.1145\/3315508.3329973"},{"key":"e_1_3_3_2_59_2","unstructured":"Hugo Touvron et\u00a0al. 2023. Llama 2: Open Foundation and Fine-tuned Chat Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.09288 (2023)."},{"key":"e_1_3_3_2_60_2","volume-title":"Proceedings of the International Conference on Machine Learning (ICML)","author":"Touvron Hugo","year":"2021","unstructured":"Hugo Touvron, Matthieu Cord, Matthijs Douze, Francisco Massa, Alexandre Sablayrolles, and Herv\u00e9 J\u00e9gou. 2021. Training data-efficient image transformers & distillation through attention. In Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"e_1_3_3_2_61_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00021"},{"key":"e_1_3_3_2_62_2","volume-title":"Proceedings of the USENIX Symposium on Operating Systems Design and Implementation (OSDI)","author":"Wang Lei","year":"2024","unstructured":"Lei Wang, Lingxiao Ma, Shijie Cao, Quanlu Zhang, Jilong Xue, Yining Shi, Ningxin Zheng, Ziming Miao, Fan Yang, Ting Cao, Yuqing Yang, and Mao Yang. 2024. Ladder: Enabling Efficient Low-Precision Deep Learning Computing through Hardware-aware Tensor Transformation. In Proceedings of the USENIX Symposium on Operating Systems Design and Implementation (OSDI)."},{"key":"e_1_3_3_2_63_2","unstructured":"Thomas Wolf Lysandre Debut Victor Sanh Julien Chaumond Clement Delangue Anthony Moi Pierric Cistac Tim Rault R\u00e9mi Louf Morgan Funtowicz et\u00a0al. 2019. HuggingFace\u2019s Transformers: State-of-the-art Natural Language Processing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1910.03771 (2019)."},{"key":"e_1_3_3_2_64_2","volume-title":"Proceedings of the International Conference on Machine Learning (ICML)","author":"Xiao Guangxuan","year":"2023","unstructured":"Guangxuan Xiao, Ji Lin, Mickael Seznec, Hao Wu, Julien Demouth, and Song Han. 2023. SmoothQuant: Accurate and Efficient Post-Training Quantization for Large Language Models. In Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"e_1_3_3_2_65_2","unstructured":"Qwen:\u00a0An Yang Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chengyuan Li Dayiheng Liu Fei Huang Haoran Wei Huan Lin Jian Yang Jianhong Tu Jianwei Zhang Jianxin Yang Jiaxi Yang Jingren Zhou Junyang Lin Kai Dang Keming Lu Keqin Bao Kexin Yang Le Yu Mei Li Mingfeng Xue Pei Zhang Qin Zhu Rui Men Runji Lin Tianhao Li Tianyi Tang Tingyu Xia Xingzhang Ren Xuancheng Ren Yang Fan Yang Su Yichang Zhang Yu Wan Yuqiong Liu Zeyu Cui Zhenru Zhang and Zihan Qiu. 2025. Qwen2.5 Technical Report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.15115 (2025)."},{"key":"e_1_3_3_2_66_2","volume-title":"Proceedings of the International Conference on Machine Learning (ICML)","author":"Yeh Thomas","year":"2022","unstructured":"Thomas Yeh, Max Sterner, Zerlina Lai, Brandon Chuang, and Alexander Ihler. 2022. Be Like Water: Adaptive Floating Point for Machine Learning. In Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"e_1_3_3_2_67_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651368"},{"key":"e_1_3_3_2_68_2","doi-asserted-by":"publisher","DOI":"10.1145\/3470496.3527438"},{"key":"e_1_3_3_2_69_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.617"},{"key":"e_1_3_3_2_70_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614309"},{"key":"e_1_3_3_2_71_2","unstructured":"Susan Zhang Stephen Roller Naman Goyal Mikel Artetxe Moya Chen Shuohui Chen Christopher Dewan Mona Diab Xian Li Xi\u00a0Victoria Lin Todor Mihaylov Myle Ott Sam Shleifer Kurt Shuster Daniel Simig Punit\u00a0Singh Koura Anjali Sridhar Tianlu Wang and Luke Zettlemoyer. 2022. OPT: Open Pre-trained Transformer Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2205.01068 (2022)."},{"key":"e_1_3_3_2_72_2","volume-title":"IEEE International Symposium on High-Performance Computer Architecture (HPCA)","author":"Zhang Sai\u00a0Qian","year":"2022","unstructured":"Sai\u00a0Qian Zhang, Bradley McDanel, and HT Kung. 2022. FAST: DNN Training Under Variable Precision Block Floating Point with Stochastic Rounding. In IEEE International Symposium on High-Performance Computer Architecture (HPCA)."},{"key":"e_1_3_3_2_73_2","volume-title":"Proceedings of the International Conference on Machine Learning (ICML)","author":"Zhao Ritchie","year":"2019","unstructured":"Ritchie Zhao, Yuwei Hu, Jordan Dotzel, Chris De\u00a0Sa, and Zhiru Zhang. 2019. Improving Neural Network Quantization Without Retraining Using Outlier Channel Splitting. In Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"e_1_3_3_2_74_2","volume-title":"Proceedings of Machine Learning and Systems (MLSys)","author":"Zhao Yilong","year":"2024","unstructured":"Yilong Zhao, Chien-Yu Lin, Kan Zhu, Zihao Ye, Lequn Chen, Size Zheng, Luis Ceze, Arvind Krishnamurthy, Tianqi Chen, and Baris Kasikci. 2024. Atom: Low-Bit Quantization for Efficient and Accurate LLM Serving. In Proceedings of Machine Learning and Systems (MLSys)."},{"key":"e_1_3_3_2_75_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00061"}],"event":{"name":"MICRO 2025: 58th IEEE\/ACM International Symposium on Microarchitecture","location":"Seoul Korea","acronym":"MICRO 2025","sponsor":["SIGMICRO ACM Special Interest Group on Microarchitectural Research and Processing"]},"container-title":["Proceedings of the 58th IEEE\/ACM International Symposium on Microarchitecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3725843.3756118","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3725843.3756118","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,26]],"date-time":"2026-01-26T21:41:59Z","timestamp":1769463719000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3725843.3756118"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,17]]},"references-count":74,"alternative-id":["10.1145\/3725843.3756118","10.1145\/3725843"],"URL":"https:\/\/doi.org\/10.1145\/3725843.3756118","relation":{},"subject":[],"published":{"date-parts":[[2025,10,17]]},"assertion":[{"value":"2025-10-17","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}