{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T05:12:26Z","timestamp":1783746746803,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":57,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T00:00:00Z","timestamp":1783900800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2025YFB3003702"],"award-info":[{"award-number":["2025YFB3003702"]}]},{"name":"Innovation Funding of ICT\u200c CAS","award":["E461050"],"award-info":[{"award-number":["E461050"]}]},{"name":"National Natural Science Foundation of China","award":["62032023\u200c T2125013"],"award-info":[{"award-number":["62032023\u200c T2125013"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,13]]},"DOI":"10.1145\/3806645.3807584","type":"proceedings-article","created":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T04:21:11Z","timestamp":1783743671000},"page":"235-248","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["TACO: Efficient Communication Compression of Intermediate Tensors for Scalable Tensor-Parallel LLM Training"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-7390-0220","authenticated-orcid":false,"given":"Man","family":"Liu","sequence":"first","affiliation":[{"name":"Hangzhou Institute for Advanced Study, University of Chinese Academy of Sciences, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9410-5365","authenticated-orcid":false,"given":"Xingchen","family":"Liu","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-1880-7886","authenticated-orcid":false,"given":"Xingjian","family":"Tian","sequence":"additional","affiliation":[{"name":"Hangzhou Institute for Advanced Study, University of Chinese Academy of Sciences, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7633-0486","authenticated-orcid":false,"given":"Bing","family":"Lu","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-3264-6074","authenticated-orcid":false,"given":"Shengkai","family":"Lyu","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2043-2456","authenticated-orcid":false,"given":"Shengquan","family":"Yin","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, Anhui, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-5541-3519","authenticated-orcid":false,"given":"Wenjing","family":"Huang","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8263-1986","authenticated-orcid":false,"given":"Zheng","family":"Wei","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-7081-5172","authenticated-orcid":false,"given":"Hairui","family":"Zhao","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6361-5948","authenticated-orcid":false,"given":"Guangming","family":"Tan","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5422-4497","authenticated-orcid":false,"given":"Dingwen","family":"Tao","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,13]]},"reference":[{"key":"e_1_3_3_1_2_2","series-title":"(NIPS\u201917)","first-page":"1707","volume-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems","author":"Alistarh Dan","year":"2017","unstructured":"Dan Alistarh, Demjan Grubic, Jerry\u00a0Z. Li, Ryota Tomioka, and Milan Vojnovic. 2017. QSGD: communication-efficient SGD via gradient quantization and encoding. In Proceedings of the 31st International Conference on Neural Information Processing Systems (Long Beach, California, USA) (NIPS\u201917). Curran Associates Inc., Red Hook, NY, USA, 1707\u20131718."},{"key":"e_1_3_3_1_3_2","unstructured":"Quentin Anthony Benjamin Michalowicz Jacob Hatef Lang Xu Mustafa Abduljabbar Aamir Shafi Hari Subramoni and Dhabaleswar Panda. 2024. Demystifying the Communication Characteristics for Distributed Transformer Models. arxiv:https:\/\/arXiv.org\/abs\/2408.10197\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2408.10197"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.197"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"crossref","unstructured":"Saleh Ashkboos Amirkeivan Mohtashami Maximilian\u00a0L Croci Bo Li Pashmina Cameron Martin Jaggi Dan Alistarh Torsten Hoefler and James Hensman. 2024. Quarot: Outlier-free 4-bit inference in rotated llms. Advances in Neural Information Processing Systems 37 (2024) 100213\u2013100240.","DOI":"10.52202\/079017-3180"},{"key":"e_1_3_3_1_6_2","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared\u00a0D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et\u00a0al. 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020) 1877\u20131901."},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"crossref","unstructured":"Jerry Chee Yaohui Cai Volodymyr Kuleshov and Christopher\u00a0M De\u00a0Sa. 2023. Quip: 2-bit quantization of large language models with guarantees. Advances in Neural Information Processing Systems 36 (2023) 4396\u20134429.","DOI":"10.52202\/075280-0196"},{"key":"e_1_3_3_1_8_2","unstructured":"Chia-Yu Chen Jiamin Ni Songtao Lu Xiaodong Cui Pin-Yu Chen Xiao Sun Naigang Wang Swagath Venkataramani Vijayalakshmi\u00a0Viji Srinivasan Wei Zhang et\u00a0al. 2020. Scalecom: Scalable sparsified gradient compression for communication-efficient distributed training. Advances in Neural Information Processing Systems 33 (2020) 13551\u201313563."},{"key":"e_1_3_3_1_9_2","unstructured":"Jianfei Chen Lianmin Zheng Zhewei Yao Dequan Wang Ion Stoica Michael\u00a0W. Mahoney and Joseph\u00a0E. Gonzalez. 2021. ActNN: Reducing Training Memory Footprint via 2-Bit Activation Compressed Training. arxiv:https:\/\/arXiv.org\/abs\/2104.14129\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2104.14129"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607037"},{"key":"e_1_3_3_1_11_2","series-title":"(ICML\u201924)","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"Chen Yanxi","year":"2024","unstructured":"Yanxi Chen, Xuchen Pan, Yaliang Li, Bolin Ding, and Jingren Zhou. 2024. EE-LLM: large-scale training and inference of early-exit large language models with 3D parallelism. In Proceedings of the 41st International Conference on Machine Learning(ICML\u201924). JMLR.org, Vienna, Austria, Article 277, 27\u00a0pages."},{"key":"e_1_3_3_1_12_2","unstructured":"Aakanksha Chowdhery Sharan Narang Jacob Devlin Maarten Bosma Gaurav Mishra Adam Roberts Paul Barham Hyung\u00a0Won Chung Charles Sutton Sebastian Gehrmann et\u00a0al. 2023. Palm: Scaling language modeling with pathways. Journal of Machine Learning Research 24 240 (2023) 1\u2013113."},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"crossref","unstructured":"Tim Dettmers Artidoro Pagnoni Ari Holtzman and Luke Zettlemoyer. 2023. Qlora: Efficient finetuning of quantized llms. Advances in neural information processing systems 36 (2023) 10088\u201310115.","DOI":"10.52202\/075280-0441"},{"key":"e_1_3_3_1_14_2","unstructured":"Harry Dong Tyler Johnson Minsik Cho and Emad Soroush. 2024. Towards Low-bit Communication for Tensor Parallel LLM Inference. arxiv:https:\/\/arXiv.org\/abs\/2411.07942\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2411.07942"},{"key":"e_1_3_3_1_15_2","unstructured":"Vage Egiazarian Roberto\u00a0L. Castro Denis Kuznedelev Andrei Panferov Eldar Kurtic Shubhra Pandit Alexandre Marques Mark Kurtz Saleh Ashkboos Torsten Hoefler and Dan Alistarh. 2026. Bridging the Gap Between Promise and Performance for Microscaling FP4 Quantization. arxiv:https:\/\/arXiv.org\/abs\/2509.23202\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2509.23202"},{"key":"e_1_3_3_1_16_2","unstructured":"Elias Frantar Saleh Ashkboos Torsten Hoefler and Dan Alistarh. 2023. GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers. arxiv:https:\/\/arXiv.org\/abs\/2210.17323\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2210.17323"},{"key":"e_1_3_3_1_17_2","unstructured":"Trevor Gale Deepak Narayanan Cliff Young and Matei Zaharia. 2022. MegaBlocks: Efficient Sparse Training with Mixture-of-Experts. arxiv:https:\/\/arXiv.org\/abs\/2211.15841\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2211.15841"},{"key":"e_1_3_3_1_18_2","unstructured":"Leo Gao Stella Biderman Sid Black Laurence Golding Travis Hoppe Charles Foster Jason Phang Horace He Anish Thite Noa Nabeshima Shawn Presser and Connor Leahy. 2020. The Pile: An 800GB Dataset of Diverse Text for Language Modeling. arxiv:https:\/\/arXiv.org\/abs\/2101.00027\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2101.00027"},{"key":"e_1_3_3_1_19_2","unstructured":"Aaron Grattafiori et\u00a0al. 2024. The Llama 3 Herd of Models. arxiv:https:\/\/arXiv.org\/abs\/2407.21783\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"e_1_3_3_1_20_2","unstructured":"Guangxin He Yuan Cao Yutong He Tianyi Bai Kun Yuan and Binhang Yuan. 2025. TAH-QUANT: Effective Activation Quantization in Pipeline Parallelism over Slow Network. arxiv:https:\/\/arXiv.org\/abs\/2506.01352\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2506.01352"},{"key":"e_1_3_3_1_21_2","volume-title":"Proceedings of the 33rd International Conference on Neural Information Processing Systems","author":"Huang Yanping","year":"2019","unstructured":"Yanping Huang, Youlong Cheng, Ankur Bapna, Orhan Firat, Mia\u00a0Xu Chen, Dehao Chen, HyoukJoong Lee, Jiquan Ngiam, Quoc\u00a0V. Le, Yonghui Wu, and Zhifeng Chen. 2019. GPipe: efficient training of giant neural networks using pipeline parallelism. In Proceedings of the 33rd International Conference on Neural Information Processing Systems. Curran Associates Inc., Red Hook, NY, USA, Article 10, 10\u00a0pages."},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"crossref","unstructured":"Jinda Jia Cong Xie Hanlin Lu Daoce Wang Hao Feng Chengming Zhang Baixi Sun Haibin Lin Zhi Zhang Xin Liu et\u00a0al. 2024. Sdp4bit: Toward 4-bit communication quantization in sharded data parallelism for LLM training. Advances in Neural Information Processing Systems 37 (2024) 8734\u20138759.","DOI":"10.52202\/079017-0279"},{"key":"e_1_3_3_1_23_2","unstructured":"Ziheng Jiang Haibin Lin Yinmin Zhong Qi Huang Yangrui Chen Zhi Zhang Yanghua Peng Xiang Li Cong Xie Shibiao Nong Yulu Jia Sun He Hongmin Chen Zhihao Bai Qi Hou Shipeng Yan Ding Zhou Yiyao Sheng Zhuo Jiang Haohan Xu Haoran Wei Zhang Zhang Pengfei Nie Leqi Zou Sida Zhao Liang Xiang Zherui Liu Zhe Li Xiaoying Jia Jianxi Ye Xin Jin and Xin Liu. 2024. MegaScale: Scaling Large Language Model Training to More Than 10 000 GPUs. arxiv:https:\/\/arXiv.org\/abs\/2402.15627\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2402.15627"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"crossref","unstructured":"Andrey Kuzmin Mart Van\u00a0Baalen Yuwei Ren Markus Nagel Jorn Peters and Tijmen Blankevoort. 2022. Fp8 quantization: The power of the exponent. Advances in Neural Information Processing Systems 35 (2022) 14651\u201314662.","DOI":"10.52202\/068431-1065"},{"key":"e_1_3_3_1_25_2","unstructured":"Itay Lamprecht Asaf Karnieli Yair Hanani Niv Giladi and Daniel Soudry. 2025. Tensor-Parallelism with Partially Synchronized Activations. NeurIPS 2025 Poster. https:\/\/openreview.net\/forum?id=fyeSq3m8CY Accepted as NeurIPS 2025 Poster."},{"key":"e_1_3_3_1_26_2","unstructured":"Qingyuan Li Bo Zhang Liang Ye Yifan Zhang Wei Wu Yerui Sun Lin Ma and Yuchen Xie. 2024. Flash Communication: Reducing Tensor Parallelization Bottleneck for Fast Large Language Model Inference. arxiv:https:\/\/arXiv.org\/abs\/2412.04964\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2412.04964"},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503221.3508399"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.134"},{"key":"e_1_3_3_1_29_2","unstructured":"Yujun Lin Song Han Huizi Mao Yu Wang and William\u00a0J. Dally. 2020. Deep Gradient Compression: Reducing the Communication Bandwidth for Distributed Training. arxiv:https:\/\/arXiv.org\/abs\/1712.01887\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1712.01887"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.1145\/3774934.3786432"},{"key":"e_1_3_3_1_31_2","unstructured":"Zechun Liu Barlas Oguz Changsheng Zhao Ernie Chang Pierre Stock Yashar Mehdad Yangyang Shi Raghuraman Krishnamoorthi and Vikas Chandra. 2023. LLM-QAT: Data-Free Quantization Aware Training for Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2305.17888\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2305.17888"},{"key":"e_1_3_3_1_32_2","series-title":"(ICML\u201923)","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Markov Ilia","year":"2023","unstructured":"Ilia Markov, Adrian Vladu, Qi Guo, and Dan Alistarh. 2023. Quantized distributed training of large models with convergence guarantees. In Proceedings of the 40th International Conference on Machine Learning (Honolulu, Hawaii, USA) (ICML\u201923). JMLR.org, Honolulu, Hawaii, USA, Article 1001, 25\u00a0pages."},{"key":"e_1_3_3_1_33_2","unstructured":"Paulius Micikevicius Sharan Narang Jonah Alben Gregory Diamos Erich Elsen David Garcia Boris Ginsburg Michael Houston Oleksii Kuchaiev Ganesh Venkatesh and Hao Wu. 2018. Mixed Precision Training. arxiv:https:\/\/arXiv.org\/abs\/1710.03740\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/1710.03740"},{"key":"e_1_3_3_1_34_2","unstructured":"Paulius Micikevicius Dusan Stosic Neil Burgess Marius Cornea Pradeep Dubey Richard Grisenthwaite Sangwon Ha Alexander Heinecke Patrick Judd John Kamalu Naveen Mellempudi Stuart Oberman Mohammad Shoeybi Michael Siu and Hao Wu. 2022. FP8 Formats for Deep Learning. arxiv:https:\/\/arXiv.org\/abs\/2209.05433\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2209.05433"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476209"},{"key":"e_1_3_3_1_37_2","unstructured":"Keiran Paster Marco\u00a0Dos Santos Zhangir Azerbayev and Jimmy Ba. 2023. OpenWebMath: An Open Dataset of High-Quality Mathematical Web Text. arxiv:https:\/\/arXiv.org\/abs\/2310.06786\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2310.06786"},{"key":"e_1_3_3_1_38_2","volume-title":"Proceedings of the 33rd International Conference on Neural Information Processing Systems","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, Alban Desmaison, Andreas K\u00f6pf, Edward Yang, Zach DeVito, Martin Raison, Alykhan Tejani, Sasank Chilamkurthy, Benoit Steiner, Lu Fang, Junjie Bai, and Soumith Chintala. 2019. PyTorch: an imperative style, high-performance deep learning library. In Proceedings of the 33rd International Conference on Neural Information Processing Systems. Curran Associates Inc., Red Hook, NY, USA, Article 721, 12\u00a0pages."},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"publisher","DOI":"10.1145\/3721146.3721946"},{"key":"e_1_3_3_1_40_2","unstructured":"Samyam Rajbhandari Jeff Rasley Olatunji Ruwase and Yuxiong He. 2020. ZeRO: Memory Optimizations Toward Training Trillion Parameter Models. arxiv:https:\/\/arXiv.org\/abs\/1910.02054\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/1910.02054"},{"key":"e_1_3_3_1_41_2","doi-asserted-by":"publisher","unstructured":"M.\u00a0I. Rudakov A.\u00a0N. Beznosikov Ya.\u00a0A. Kholodov and A.\u00a0V. Gasnikov. 2023. Activations and Gradients Compression for Model-Parallel Training. Doklady Mathematics 108 S2 (Dec. 2023) S272\u2013S281. 10.1134\/s1064562423701314","DOI":"10.1134\/s1064562423701314"},{"key":"e_1_3_3_1_42_2","unstructured":"Semyon Savkin. 2025. Quantization Methods for Matrix Multiplication and Efficient Transformers. Ph.\u00a0D. Dissertation. MASSACHUSETTS INSTITUTE OF TECHNOLOGY."},{"key":"e_1_3_3_1_43_2","unstructured":"Haihao Shen Naveen Mellempudi Xin He Qun Gao Chang Wang and Mengni Wang. 2024. Efficient post-training quantization with fp8 formats. Proceedings of Machine Learning and Systems 6 (2024) 483\u2013498."},{"key":"e_1_3_3_1_44_2","unstructured":"Mohammad Shoeybi Mostofa Patwary Raul Puri Patrick LeGresley Jared Casper and Bryan Catanzaro. 2020. Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. arxiv:https:\/\/arXiv.org\/abs\/1909.08053\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/1909.08053"},{"key":"e_1_3_3_1_45_2","unstructured":"Shaden Smith Mostofa Patwary Brandon Norick Patrick LeGresley Samyam Rajbhandari Jared Casper Zhun Liu Shrimai Prabhumoye George Zerveas Vijay Korthikanti Elton Zhang Rewon Child Reza\u00a0Yazdani Aminabadi Julie Bernauer Xia Song Mohammad Shoeybi Yuxiong He Michael Houston Saurabh Tiwary and Bryan Catanzaro. 2022. Using DeepSpeed and Megatron to Train Megatron-Turing NLG 530B A Large-Scale Generative Language Model. arxiv:https:\/\/arXiv.org\/abs\/2201.11990\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2201.11990"},{"key":"e_1_3_3_1_46_2","unstructured":"Yuxuan Sun Ruikang Liu Haoli Bai Han Bao Kang Zhao Yuening Li Jiaxin Hu Xianzhi Yu Lu Hou Chun Yuan Xin Jiang Wulong Liu and Jun Yao. 2025. FlatQuant: Flatness Matters for LLM Quantization. arxiv:https:\/\/arXiv.org\/abs\/2410.09426\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2410.09426"},{"key":"e_1_3_3_1_47_2","unstructured":"Qwen Team. 2024. Qwen2.5: A Party of Foundation Models. https:\/\/qwenlm.github.io\/blog\/qwen2.5\/"},{"key":"e_1_3_3_1_48_2","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard Xavier Martinet Marie-Anne Lachaux Timoth\u00e9e Lacroix Baptiste Rozi\u00e8re Naman Goyal Eric Hambro Faisal Azhar Aurelien Rodriguez Armand Joulin Edouard Grave and Guillaume Lample. 2023. LLaMA: Open and Efficient Foundation Language Models. arxiv:https:\/\/arXiv.org\/abs\/2302.13971\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2302.13971"},{"key":"e_1_3_3_1_49_2","series-title":"(ICML\u201924)","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"Tseng Albert","year":"2024","unstructured":"Albert Tseng, Jerry Chee, Qingyao Sun, Volodymyr Kuleshov, and Christopher De\u00a0Sa. 2024. QuIP#: even better LLM quantization with hadamard incoherence and lattice codebooks. In Proceedings of the 41st International Conference on Machine Learning (Vienna, Austria) (ICML\u201924). JMLR.org, Vienna, Austria, Article 1987, 27\u00a0pages."},{"key":"e_1_3_3_1_50_2","volume-title":"Proceedings of the 33rd International Conference on Neural Information Processing Systems","author":"Vogels Thijs","year":"2019","unstructured":"Thijs Vogels, Sai\u00a0Praneeth Karimireddy, and Martin Jaggi. 2019. PowerSGD: practical low-rank gradient compression for distributed optimization. In Proceedings of the 33rd International Conference on Neural Information Processing Systems. Curran Associates Inc., Red Hook, NY, USA, Article 1278, 10\u00a0pages."},{"key":"e_1_3_3_1_51_2","unstructured":"Haiquan Wang Chaoyi Ruan Jia He Jiaqi Ruan Chengjie Tang Xiaosong Ma and Cheng Li. 2024. Hiding Communication Cost in Distributed LLM Training via Micro-batch Co-execution. arxiv:https:\/\/arXiv.org\/abs\/2411.15871\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2411.15871"},{"key":"e_1_3_3_1_52_2","doi-asserted-by":"crossref","unstructured":"Jue Wang Binhang Yuan Luka Rimanic Yongjun He Tri Dao Beidi Chen Christopher R\u00e9 and Ce Zhang. 2022. Fine-tuning language models over slow networks using activation quantization with guarantees. Advances in Neural Information Processing Systems 35 (2022) 19215\u201319230.","DOI":"10.52202\/068431-1397"},{"key":"e_1_3_3_1_53_2","unstructured":"BigScience Workshop Teven\u00a0Le Scao Angela Fan Christopher Akiki et\u00a0al. 2023. BLOOM: A 176B-Parameter Open-Access Multilingual Language Model. arxiv:https:\/\/arXiv.org\/abs\/2211.05100\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2211.05100"},{"key":"e_1_3_3_1_54_2","unstructured":"Guangxuan Xiao Ji Lin Mickael Seznec Hao Wu Julien Demouth and Song Han. 2024. SmoothQuant: Accurate and Efficient Post-Training Quantization for Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2211.10438\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2211.10438"},{"key":"e_1_3_3_1_55_2","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid59990.2024.00031"},{"key":"e_1_3_3_1_56_2","unstructured":"Qingao Yi Jiaang Duan Hanwen Hu Qin Hua Haiyan Zhao Shiyou Qian Dingyu Yang Jian Cao Jinghua Tang Yinghao Yu Chenzhi Liao Kangjin Wang and Liping Zhang. 2025. EDGC: Entropy-driven Dynamic Gradient Compression for Efficient LLM Training. arxiv:https:\/\/arXiv.org\/abs\/2511.10333\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2511.10333"},{"key":"e_1_3_3_1_57_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS57875.2023.00031"},{"key":"e_1_3_3_1_58_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446916"}],"event":{"name":"HPDC '26: 35th International Symposium on High-Performance Parallel and Distributed Computing","location":"Cleveland USA","acronym":"HPDC '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 35th International Symposium on High-Performance Parallel and Distributed Computing"],"original-title":[],"deposited":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T04:24:58Z","timestamp":1783743898000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3806645.3807584"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,13]]},"references-count":57,"alternative-id":["10.1145\/3806645.3807584","10.1145\/3806645"],"URL":"https:\/\/doi.org\/10.1145\/3806645.3807584","relation":{},"subject":[],"published":{"date-parts":[[2026,7,13]]},"assertion":[{"value":"2026-07-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}