{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T00:07:50Z","timestamp":1755907670226,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":77,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,11,18]],"date-time":"2024-11-18T00:00:00Z","timestamp":1731888000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,11,18]]},"DOI":"10.1145\/3696348.3696893","type":"proceedings-article","created":{"date-parts":[[2024,11,11]],"date-time":"2024-11-11T00:20:52Z","timestamp":1731284452000},"page":"195-204","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["I've Got 99 Problems But FLOPS Ain't One"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-3118-4617","authenticated-orcid":false,"given":"Alexandru M.","family":"Gherghescu","sequence":"first","affiliation":[{"name":"University Politehnica of Bucharest"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2890-7980","authenticated-orcid":false,"given":"Vlad-Andrei","family":"B\u0103doiu","sequence":"additional","affiliation":[{"name":"University Politehnica of Bucharest"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4728-0404","authenticated-orcid":false,"given":"Alexandru","family":"Agache","sequence":"additional","affiliation":[{"name":"University Politehnica of Bucharest"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3202-6110","authenticated-orcid":false,"given":"Mihai-Valentin","family":"Dumitru","sequence":"additional","affiliation":[{"name":"University Politehnica of Bucharest"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0947-7000","authenticated-orcid":false,"given":"Iuliu","family":"Vasilescu","sequence":"additional","affiliation":[{"name":"University Politehnica of Bucharest"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6157-5914","authenticated-orcid":false,"given":"Radu","family":"Mantu","sequence":"additional","affiliation":[{"name":"University Politehnica of Bucharest"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5937-2162","authenticated-orcid":false,"given":"Costin","family":"Raiciu","sequence":"additional","affiliation":[{"name":"University Politehnica of Bucharest and Broadcom Inc."}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,11,18]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2019.00103"},{"volume-title":"Commodity Data Center Network Architecture","author":"Al-Fares Mohammad","key":"e_1_3_2_1_2_1","unstructured":"Mohammad Al-Fares, Alexander Loukissas, and Amin Vahdat. 2008. A Scalable, Commodity Data Center Network Architecture. In Special Interest Group on Data Communication (SIGCOMM). ACM."},{"key":"e_1_3_2_1_3_1","volume-title":"Hedera: Dynamic Flow Scheduling for Data Center Networks. In Networked Systems Design and Implementation (NSDI)","author":"Al-Fares Mohammad","year":"2010","unstructured":"Mohammad Al-Fares, Sivasankar Radhakrishnan, Barath Raghavan, Nelson Huang, and Amin Vahdat. 2010. Hedera: Dynamic Flow Scheduling for Data Center Networks. In Networked Systems Design and Implementation (NSDI). USENIX Association."},{"key":"e_1_3_2_1_4_1","unstructured":"AMD. 2020. AMD Infinity Fabric\u2122 Link. Retrieved 2024-10-13 from https:\/\/www.amd.com\/content\/dam\/amd\/en\/documents\/instinct-tech-docs\/other\/56978.pdf"},{"key":"e_1_3_2_1_5_1","volume-title":"Longformer: The long-document transformer. arXiv preprint arXiv:2004.05150","author":"Beltagy Iz","year":"2020","unstructured":"Iz Beltagy, Matthew E Peters, and Arman Cohan. 2020. Longformer: The long-document transformer. arXiv preprint arXiv:2004.05150 (2020)."},{"key":"e_1_3_2_1_6_1","unstructured":"Bloomberg. 2024. OpenAI Pitched White House on Unprecedented Data Center Buildout. Retrieved 2024-10-14 from https:\/\/www.bloomberg.com\/news\/articles\/2024-09-24\/openai-pitched-white-house-on-unprecedented-data-center-buildout"},{"key":"e_1_3_2_1_7_1","volume-title":"Rong Pan, Yanfang Le, Costin Raiciu, Mark Handley, Timo Schneider, Nils Blach, Ahmad Ghalayini, Daniel Alves, Michael Papamichael, Adrian Caulfield, and Torsten Hoefler.","author":"Bonato Tommaso","year":"2024","unstructured":"Tommaso Bonato, Abdul Kabbani, Daniele De Sensi, Rong Pan, Yanfang Le, Costin Raiciu, Mark Handley, Timo Schneider, Nils Blach, Ahmad Ghalayini, Daniel Alves, Michael Papamichael, Adrian Caulfield, and Torsten Hoefler. 2024. SMaRTT-REPS: Sender-based Marked Rapidly-adapting Trimmed & Timed Transport with Recycled Entropies. arXiv preprint arXiv:2404.01630v1 (2024). https:\/\/arxiv.org\/html\/2404.01630v1"},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems. 1877--1901","author":"Brown Tom B","year":"2020","unstructured":"Tom B Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, et al. 2020. Language models are few-shot learners. In Proceedings of the 34th International Conference on Neural Information Processing Systems. 1877--1901."},{"key":"e_1_3_2_1_9_1","unstructured":"C. Dahl R. Cui D. Cui C. Squire S. Kennedy. 2022. Pathways to achieving a 2030 coal phase-out in the United States."},{"key":"e_1_3_2_1_10_1","volume-title":"Longlora: Efficient fine-tuning of long-context large language models. arXiv preprint arXiv:2309.12307","author":"Chen Yukang","year":"2023","unstructured":"Yukang Chen, Shengju Qian, Haotian Tang, Xin Lai, Zhijian Liu, Song Han, and Jiaya Jia. 2023. Longlora: Efficient fine-tuning of long-context large language models. arXiv preprint arXiv:2309.12307 (2023)."},{"key":"e_1_3_2_1_11_1","first-page":"1","article-title":"Palm: Scaling language modeling with pathways","volume":"24","author":"Chowdhery Aakanksha","year":"2023","unstructured":"Aakanksha Chowdhery, Sharan Narang, Jacob Devlin, Maarten Bosma, Gaurav Mishra, Adam Roberts, Paul Barham, Hyung Won Chung, Charles Sutton, Sebastian Gehrmann, et al. 2023. Palm: Scaling language modeling with pathways. Journal of Machine Learning Research 24, 240 (2023), 1--113.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_12_1","volume-title":"Using Multirail Networks in High-Performance Clusters. (08","author":"Coll Salvador","year":"2002","unstructured":"Salvador Coll, Eitan Frachtenberg, Fabrizio Petrini, Adolfy Hoisie, and Leonid Gurvits. 2002. Using Multirail Networks in High-Performance Clusters. (08 2002)."},{"key":"e_1_3_2_1_13_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_1_14_1","volume-title":"Chengruidong Zhang, Yuanyuan Xu, Ning Shang, Jiahang Xu, Fan Yang, and Mao Yang.","author":"Ding Yiran","year":"2024","unstructured":"Yiran Ding, Li Lyna Zhang, Chengruidong Zhang, Yuanyuan Xu, Ning Shang, Jiahang Xu, Fan Yang, and Mao Yang. 2024. Longrope: Extending llm context window beyond 2 million tokens. arXiv preprint arXiv:2402.13753 (2024)."},{"key":"e_1_3_2_1_15_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_16_1","unstructured":"Dylan Patel and Daniel Nishball and Jeremie Eliahou Ontiveros. 2024. Multi-Datacenter Training: OpenAI's Ambitious Plan To Beat Google's Infrastructure. Retrieved 2024-10-14 from https:\/\/www.semianalysis.com\/p\/multi-datacenter-training-openais"},{"key":"e_1_3_2_1_17_1","first-page":"1","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus William","year":"2022","unstructured":"William Fedus, Barret Zoph, and Noam Shazeer. 2022. Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity. Journal of Machine Learning Research 23, 120 (2022), 1--39.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_18_1","unstructured":"Maxim Fishman Brian Chmiel Ron Banner and Daniel Soudry. 2024. Scaling FP8 training to trillion-token LLMs. arXiv:2409.12517 [cs.LG] https:\/\/arxiv.org\/abs\/2409.12517"},{"key":"e_1_3_2_1_19_1","volume-title":"Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752","author":"Gu Albert","year":"2023","unstructured":"Albert Gu and Tri Dao. 2023. Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752 (2023)."},{"volume-title":"RDMA over Commodity Ethernet at Scale","author":"Guo Chuanxiong","key":"e_1_3_2_1_20_1","unstructured":"Chuanxiong Guo, Haitao Wu, Zhong Deng, Gaurav Soni, Jianxi Ye, Jitu Padhye, and Marina Lipshteyn. 2016. RDMA over Commodity Ethernet at Scale. In Special Interest Group on Data Communication (SIGCOMM). ACM."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3555050.3569141"},{"key":"e_1_3_2_1_22_1","volume-title":"Datacenter Summit","author":"Halani Jigar","year":"2024","unstructured":"Jigar Halani. 2024. The Evolution of Data Centers: How AI Factories Are Leading the Charge!. In Datacenter Summit 2024. https:\/\/dcs.greenbusinesscentre.com\/dcs2024presentations\/1.pdf"},{"volume-title":"Re-architecting Datacenter Networks and Stacks for Low Latency and High Performance","author":"Handley Mark","key":"e_1_3_2_1_23_1","unstructured":"Mark Handley, Costin Raiciu, Alexandru Agache, Andrei Voinescu, Andrew W. Moore, Gianni Antichi, and Marcin W\u00f3jcik. 2017. Re-architecting Datacenter Networks and Stacks for Low Latency and High Performance. In Special Interest Group on Data Communication (SIGCOMM). ACM."},{"key":"e_1_3_2_1_24_1","volume-title":"Gaussian error linear units (GELUs). arXiv preprint arXiv:1606.08415","author":"Hendrycks Dan","year":"2016","unstructured":"Dan Hendrycks and Kevin Gimpel. 2016. Gaussian error linear units (GELUs). arXiv preprint arXiv:1606.08415 (2016)."},{"key":"e_1_3_2_1_25_1","unstructured":"Jordan Hoffmann Sebastian Borgeaud Arthur Mensch E Buchatskaya T Cai E Rutherford DdL Casas LA Hendricks J Welbl A Clark et al. 2022. Training compute-optimal large language models. arXiv 2022. arXiv preprint arXiv:2203.15556 10 (2022)."},{"key":"e_1_3_2_1_26_1","volume-title":"Keynote. In GPU Technology Conference","author":"Huang Jen-Hsun","year":"2024","unstructured":"Jen-Hsun Huang. 2024. Keynote. In GPU Technology Conference 2024. https:\/\/www.youtube.com\/watch?v=Y2F8yisiS6E"},{"key":"e_1_3_2_1_27_1","unstructured":"Intel. 2024. Intel Gaudi 3 AI Accelerator White Paper. Retrieved 2024-10-13 from https:\/\/www.intel.com\/content\/www\/us\/en\/content-details\/817486\/intel-gaudi-3-ai-accelerator-white-paper.html"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/2534169.2486019"},{"key":"e_1_3_2_1_29_1","unstructured":"Jarred Walton. 2024. Nvidia's next-gen AI GPU is 4X faster than Hopper: Blackwell B200 GPU delivers up to 20 petaflops of compute and other massive improvements. Retrieved 2024-10-01 from https:\/\/www.tomshardware.com\/pc-components\/gpus\/nvidias-next-gen-ai-gpu-revealed-blackwell-b200-gpu-delivers-up-to-20-petaflops-of-compute-and-massive-improvements-over-hopper-h100"},{"key":"e_1_3_2_1_30_1","volume-title":"Diego de las Casas, Emma Bou Hanna, Florian Bressand, et al.","author":"Jiang Albert Q","year":"2024","unstructured":"Albert Q Jiang, Alexandre Sablayrolles, Antoine Roux, Arthur Mensch, Blanche Savary, Chris Bamford, Devendra Singh Chaplot, Diego de las Casas, Emma Bou Hanna, Florian Bressand, et al. 2024. Mixtral of experts. arXiv preprint arXiv:2401.04088 (2024)."},{"key":"e_1_3_2_1_31_1","volume-title":"21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24)","author":"Jiang Ziheng","year":"2024","unstructured":"Ziheng Jiang, Haibin Lin, Yinmin Zhong, Qi Huang, Yangrui Chen, Zhi Zhang, Yanghua Peng, Xiang Li, Cong Xie, Shibiao Nong, et al. 2024. {MegaScale}: Scaling large language model training to more than 10,000 {GPUs}. In 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24). 745--760."},{"key":"e_1_3_2_1_32_1","volume-title":"Scaling laws for neural language models. arXiv preprint arXiv:2001.08361","author":"Kaplan Jared","year":"2020","unstructured":"Jared Kaplan, Sam McCandlish, Tom Henighan, Tom B Brown, Benjamin Chess, Rewon Child, Scott Gray, Alec Radford, Jeffrey Wu, and Dario Amodei. 2020. Scaling laws for neural language models. arXiv preprint arXiv:2001.08361 (2020)."},{"key":"e_1_3_2_1_33_1","first-page":"341","article-title":"Reducing activation recomputation in large transformer models","volume":"5","author":"Korthikanti Vijay Anand","year":"2023","unstructured":"Vijay Anand Korthikanti, Jared Casper, Sangkug Lym, Lawrence McAfee, Michael Andersch, Mohammad Shoeybi, and Bryan Catanzaro. 2023. Reducing activation recomputation in large transformer models. Proceedings of Machine Learning and Systems 5 (2023), 341--353.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_34_1","volume-title":"The depth-to-width interplay in self-attention. arXiv preprint arXiv:2006.12467","author":"Levine Yoav","year":"2020","unstructured":"Yoav Levine, Noam Wies, Or Sharir, Hofit Bata, and Amnon Shashua. 2020. The depth-to-width interplay in self-attention. arXiv preprint arXiv:2006.12467 (2020)."},{"key":"e_1_3_2_1_35_1","volume-title":"Bart: Denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. arXiv preprint arXiv:1910.13461","author":"Lewis M","year":"2019","unstructured":"M Lewis. 2019. Bart: Denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. arXiv preprint arXiv:1910.13461 (2019)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341302.3342085"},{"key":"e_1_3_2_1_37_1","volume-title":"Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu. 2019. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692 (2019)."},{"key":"e_1_3_2_1_38_1","volume-title":"The era of 1-bit llms: All large language models are in 1.58 bits. arXiv preprint arXiv:2402.17764","author":"Ma Shuming","year":"2024","unstructured":"Shuming Ma, Hongyu Wang, Lingxiao Ma, Lei Wang, Wenhui Wang, Shaohan Huang, Li Dong, Ruiping Wang, Jilong Xue, and Furu Wei. 2024. The era of 1-bit llms: All large language models are in 1.58 bits. arXiv preprint arXiv:2402.17764 (2024)."},{"key":"e_1_3_2_1_39_1","unstructured":"Mark Nossokoff and Tom Sorensen - Hyperion Research. 2024. UALink Group Formed to Develop Standard High Speed Low Latency Interconnects for Accelerated HPC and AI Systems. Retrieved 2024-10-13 from https:\/\/hyperionresearch.com\/wp-content\/uploads\/2024\/07\/Hyperion-Research-Special-Analysis-UALink-Group-Created-for-Accelerator-Interconnect-July-2024.pdf"},{"volume-title":"Measuring energy and water efficiency for Microsoft datacenters - Microsoft Datacenters. https:\/\/datacenters.microsoft.com\/sustainability\/efficiency\/","year":"2024","key":"e_1_3_2_1_40_1","unstructured":"Microsoft. 2024. Measuring energy and water efficiency for Microsoft datacenters - Microsoft Datacenters. https:\/\/datacenters.microsoft.com\/sustainability\/efficiency\/ 08 October 2024."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3230543.3230564"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476209"},{"key":"e_1_3_2_1_43_1","unstructured":"NVIDIA. 2024. Nvidia reveals Blackwell B200 GPU the 'world's most powerful chip' for AI. Retrieved 2024-10-01 from https:\/\/www.nvidia.com\/en-us\/data-center\/gb200-nvl72\/"},{"key":"e_1_3_2_1_44_1","unstructured":"Nvidia. 2024. NVLink and NVLink Switch. Retrieved 2024-10-13 from https:\/\/www.nvidia.com\/en-us\/data-center\/nvlink\/"},{"key":"e_1_3_2_1_45_1","volume-title":"Rwkv: Reinventing rnns for the transformer era. arXiv preprint arXiv:2305.13048","author":"Peng Bo","year":"2023","unstructured":"Bo Peng, Eric Alcaide, Quentin Anthony, Alon Albalak, Samuel Arcadinho, Stella Biderman, Huanqi Cao, Xin Cheng, Michael Chung, Matteo Grella, et al. 2023. Rwkv: Reinventing rnns for the transformer era. arXiv preprint arXiv:2305.13048 (2023)."},{"key":"e_1_3_2_1_46_1","volume-title":"Yarn: Efficient context window extension of large language models. arXiv preprint arXiv:2309.00071","author":"Peng Bowen","year":"2023","unstructured":"Bowen Peng, Jeffrey Quesnelle, Honglu Fan, and Enrico Shippole. 2023. Yarn: Efficient context window extension of large language models. arXiv preprint arXiv:2309.00071 (2023)."},{"key":"e_1_3_2_1_47_1","unstructured":"Jack W Rae Sebastian Borgeaud Trevor Cai Katie Millican Jordan Hoffmann Francis Song John Aslanides Sarah Henderson Roman Ring Susannah Young et al. 2021. Scaling language models: Methods analysis & insights from training gopher. arXiv preprint arXiv:2112.11446 (2021)."},{"key":"e_1_3_2_1_48_1","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter J Liu. 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. Journal of machine learning research 21, 140 (2020), 1--67.","journal-title":"Journal of machine learning research"},{"volume-title":"Improving Datacenter Performance and Robustness with Multipath TCP","author":"Raiciu Costin","key":"e_1_3_2_1_49_1","unstructured":"Costin Raiciu, Sebastien Barre, Christopher Pluntke, Adam Greenhalgh, Damon Wischik, and Mark Handley. 2010. Improving Datacenter Performance and Robustness with Multipath TCP. In Special Interest Group on Data Communication (SIGCOMM). ACM."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476205"},{"key":"e_1_3_2_1_52_1","volume-title":"2021 USENIX Annual Technical Conference (USENIX ATC 21)","author":"Ren Jie","year":"2021","unstructured":"Jie Ren, Samyam Rajbhandari, Reza Yazdani Aminabadi, Olatunji Ruwase, Shuangyan Yang, Minjia Zhang, Dong Li, and Yuxiong He. 2021. {Zero-offload}: Democratizing {billion-scale} model training. In 2021 USENIX Annual Technical Conference (USENIX ATC 21). 551--564."},{"key":"e_1_3_2_1_53_1","unstructured":"Reuters. 2024. Microsoft OpenAI plan $100 billion data-center project media report says. Retrieved 2024-10-07 from https:\/\/www.reuters.com\/technology\/microsoft-openai-planning-100-billion-data-center-project-information-reports-2024-03-29\/"},{"key":"e_1_3_2_1_54_1","unstructured":"Ryan Smith. 2024. NVIDIA Blackwell Architecture and B200\/B100 Accelerators Announced: Going Bigger With Smaller Data. Retrieved 2024-10-01 from https:\/\/www.anandtech.com\/show\/21310\/nvidia-blackwell-architecture-and-b200b100-accelerators-announced-going-bigger-with-smaller-data"},{"key":"e_1_3_2_1_55_1","unstructured":"Maximilian Schreiner.2024. GPT-4 architecture datasets costs and more leaked. Retrieved 2024-10-07 from https:\/\/the-decoder.com\/gpt-4-architecture-datasets-costs-and-more-leaked\/"},{"key":"e_1_3_2_1_56_1","unstructured":"Sean Hollister. 2024. Nvidia reveals Blackwell B200 GPU the 'world's most powerful chip' for AI. Retrieved 2024-10-01 from https:\/\/www.tomshardware.com\/pc-components\/gpus\/nvidias-next-gen-ai-gpu-revealed-blackwell-b200-gpu-delivers-up-to-20-petaflops-of-compute-and-massive-improvements-over-hopper-h100"},{"key":"e_1_3_2_1_57_1","volume-title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer. arXiv preprint arXiv:1701.06538","author":"Shazeer Noam","year":"2017","unstructured":"Noam Shazeer, Azalia Mirhoseini, Krzysztof Maziarz, Andy Davis, Quoc Le, Geoffrey Hinton, and Jeff Dean. 2017. Outrageously large neural networks: The sparsely-gated mixture-of-experts layer. arXiv preprint arXiv:1701.06538 (2017)."},{"key":"e_1_3_2_1_58_1","volume-title":"Megatron-lm: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053","author":"Shoeybi Mohammad","year":"2019","unstructured":"Mohammad Shoeybi, Mostofa Patwary, Raul Puri, Patrick LeGresley, Jared Casper, and Bryan Catanzaro. 2019. Megatron-lm: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053 (2019)."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/2785956.2787508"},{"key":"e_1_3_2_1_60_1","unstructured":"Shaden Smith Mostofa Patwary Brandon Norick Patrick LeGresley Samyam Rajbhandari Jared Casper Zhun Liu Shrimai Prabhumoye George Zerveas Vijay Korthikanti et al. 2022. Using deepspeed and megatron to train megatron-turing nlg 530b a large-scale generative language model. arXiv preprint arXiv:2201.11990 (2022)."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.127063"},{"key":"e_1_3_2_1_62_1","first-page":"1796","article-title":"Ultra-low precision 4-bit training of deep neural networks","volume":"33","author":"Sun Xiao","year":"2020","unstructured":"Xiao Sun, Naigang Wang, Chia-Yu Chen, Jiamin Ni, Ankur Agrawal, Xiaodong Cui, Swagath Venkataramani, Kaoutar El Maghraoui, Vijayalakshmi Viji Srinivasan, and Kailash Gopalakrishnan. 2020. Ultra-low precision 4-bit training of deep neural networks. Advances in Neural Information Processing Systems 33 (2020), 1796--1807.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_63_1","volume-title":"Retentive network: A successor to transformer for large language models. arXiv preprint arXiv:2307.08621","author":"Sun Yutao","year":"2023","unstructured":"Yutao Sun, Li Dong, Shaohan Huang, Shuming Ma, Yuqing Xia, Jilong Xue, Jianyong Wang, and Furu Wei. 2023. Retentive network: A successor to transformer for large language models. arXiv preprint arXiv:2307.08621 (2023)."},{"key":"e_1_3_2_1_64_1","volume-title":"Juliette Love, et al.","author":"Team Gemma","year":"2024","unstructured":"Gemma Team, Thomas Mesnard, Cassidy Hardin, Robert Dadashi, Surya Bhupatiraju, Shreya Pathak, Laurent Sifre, Morgane Rivi\u00e8re, Mihir Sanjay Kale, Juliette Love, et al. 2024. Gemma: Open models based on gemini research and technology. arXiv preprint arXiv:2403.08295 (2024)."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/HCS61935.2024.10665247"},{"key":"e_1_3_2_1_66_1","unstructured":"Tom's Hardware. 2024. China makes AI breakthrough reportedly trains generative AI model across multiple data centers and GPU architectures. Retrieved 2024-10-14 from https:\/\/www.tomshardware.com\/techindustry\/artificial-intelligence\/china-makes-ai-breakthrough-reportedly-trains-generative-ai-model-across-multiple-datacenters-and-gpu-architectures"},{"key":"e_1_3_2_1_67_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al. 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_68_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_1_69_1","unstructured":"Ultra Ethernet Consortium. 2023. Overview of and Motivation for the Forthcoming Ultra Ethernet Consortium Specification. Retrieved 2024-10-14 from https:\/\/ultraethernet.org\/wp-content\/uploads\/sites\/20\/2023\/10\/23.07.12-UEC-1.0-Overview-FINAL-WITH-LOGO.pdf"},{"key":"e_1_3_2_1_70_1","volume-title":"summer nuclear outages rose","author":"U.S. Energy Information Administration (EIA). 2023. U.S.","year":"2023","unstructured":"U.S. Energy Information Administration (EIA). 2023. U.S. summer nuclear outages rose in 2023, returning to 2021 level. https:\/\/www.eia.gov\/todayinenergy\/detail.php?id=60682# 08 October 2024."},{"key":"e_1_3_2_1_71_1","volume-title":"https:\/\/www.eia.gov\/electricity\/gridmonitor\/dashboard\/electric_overview\/US48\/US48","author":"U.S. Energy Information Administration (EIA). 2024. U.S. Electricity Overview (U.S. Lower 48).","year":"2024","unstructured":"U.S. Energy Information Administration (EIA). 2024. U.S. Electricity Overview (U.S. Lower 48). https:\/\/www.eia.gov\/electricity\/gridmonitor\/dashboard\/electric_overview\/US48\/US48 08 October 2024."},{"key":"e_1_3_2_1_72_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_73_1","volume-title":"Bitnet: Scaling 1-bit transformers for large language models. arXiv preprint arXiv:2310.11453","author":"Wang Hongyu","year":"2023","unstructured":"Hongyu Wang, Shuming Ma, Li Dong, Shaohan Huang, Huaijie Wang, Lingxiao Ma, Fan Yang, Ruiping Wang, Yi Wu, and Furu Wei. 2023. Bitnet: Scaling 1-bit transformers for large language models. arXiv preprint arXiv:2310.11453 (2023)."},{"key":"e_1_3_2_1_74_1","unstructured":"Weissberger Alan. 2024. AI adoption to accelerate growth in the $215 billion Data Center market. https:\/\/techblog.comsoc.org\/2024\/09\/15\/bofa-ai-adoption-to-accelerate-growth-in-the-215-billion-data-center-market\/"},{"key":"e_1_3_2_1_75_1","volume-title":"Soaring from 4K to 400K: Extending LLM's Context with Activation Beacon. arXiv preprint arXiv:2401.03462","author":"Zhang Peitian","year":"2024","unstructured":"Peitian Zhang, Zheng Liu, Shitao Xiao, Ninglu Shao, Qiwei Ye, and Zhicheng Dou. 2024. Soaring from 4K to 400K: Extending LLM's Context with Activation Beacon. arXiv preprint arXiv:2401.03462 (2024)."},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"crossref","unstructured":"Yanli Zhao Andrew Gu Rohan Varma Liang Luo Chien-Chin Huang Min Xu Less Wright Hamid Shojanazeri Myle Ott Sam Shleifer et al. 2023. Pytorch fsdp: experiences on scaling fully sharded data parallel. arXiv preprint arXiv:2304.11277 (2023).","DOI":"10.14778\/3611540.3611569"},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.1145\/2785956.2787484"}],"event":{"name":"HotNets '24: The 23rd ACM Workshop on Hot Topics in Networks","sponsor":["SIGCOMM ACM Special Interest Group on Data Communication"],"location":"Irvine CA USA","acronym":"HotNets '24"},"container-title":["Proceedings of the 23rd ACM Workshop on Hot Topics in Networks"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696348.3696893","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3696348.3696893","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T16:07:00Z","timestamp":1755878820000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696348.3696893"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,18]]},"references-count":77,"alternative-id":["10.1145\/3696348.3696893","10.1145\/3696348"],"URL":"https:\/\/doi.org\/10.1145\/3696348.3696893","relation":{},"subject":[],"published":{"date-parts":[[2024,11,18]]},"assertion":[{"value":"2024-11-18","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}