{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T01:43:19Z","timestamp":1787017399621,"version":"build-2736575974"},"publisher-location":"New York, NY, USA","reference-count":65,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,4,27]],"date-time":"2024-04-27T00:00:00Z","timestamp":1714176000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["RS-2023-00222663"],"award-info":[{"award-number":["RS-2023-00222663"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Institute for Information & communications Technology Promotion of Korea","award":["2018-0-00581"],"award-info":[{"award-number":["2018-0-00581"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,4,27]]},"DOI":"10.1145\/3620666.3651362","type":"proceedings-article","created":{"date-parts":[[2024,4,24]],"date-time":"2024-04-24T08:08:21Z","timestamp":1713946101000},"page":"999-1015","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":25,"title":["TCCL: Discovering Better Communication Paths for PCIe GPU Clusters"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5392-4413","authenticated-orcid":false,"given":"Heehoon","family":"Kim","sequence":"first","affiliation":[{"name":"Seoul National University, Seoul, Korea, South ? Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-8788-7405","authenticated-orcid":false,"given":"Junyeol","family":"Ryu","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4638-8170","authenticated-orcid":false,"given":"Jaejin","family":"Lee","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Korea, South ? Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,4,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3563766.3564110"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33518-1_16"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441620"},{"key":"e_1_3_2_1_4_1","volume-title":"Alibaba machine learning platform for AI. https:\/\/www.alibabacloud.com\/product\/machine-learning","author":"Cloud Alibaba","year":"2023","unstructured":"Alibaba Cloud. Alibaba machine learning platform for AI. https:\/\/www.alibabacloud.com\/product\/machine-learning, 2023."},{"key":"e_1_3_2_1_5_1","volume-title":"GC3: An optimizing compiler for GPU collective communication. arXiv preprint arXiv:2201.11840","author":"Cowan Meghan","year":"2022","unstructured":"Meghan Cowan, Saeed Maleki, Madanlal Musuvathi, Olli Saarikivi, and Yifan Xiong. GC3: An optimizing compiler for GPU collective communication. arXiv preprint arXiv:2201.11840, 2022."},{"key":"e_1_3_2_1_6_1","volume-title":"BERT: Pre-training of deep bidirectional Transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. BERT: Pre-training of deep bidirectional Transformers for language understanding. arXiv preprint arXiv:1810.04805, 2018."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544585.3544600"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.5555\/3586589.3586709"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-30218-6_19"},{"key":"e_1_3_2_1_10_1","first-page":"829","article-title":"In-network aggregation for shared machine learning clusters","volume":"3","author":"Gebara Nadeen","year":"2021","unstructured":"Nadeen Gebara, Manya Ghobadi, and Paolo Costa. In-network aggregation for shared machine learning clusters. Proceedings of Machine Learning and Systems, 3:829--844, 2021.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_11_1","volume-title":"Google Cloud Vertex AI. https:\/\/cloud.google.com\/vertex-ai","year":"2023","unstructured":"Google. Google Cloud Vertex AI. https:\/\/cloud.google.com\/vertex-ai, 2023."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-8191(06)80021-9"},{"key":"e_1_3_2_1_13_1","unstructured":"Yanping Huang Youlong Cheng Ankur Bapna Orhan Firat Mia Xu Chen Dehao Chen HyoukJoong Lee Jiquan Ngiam Quoc V. Le Yonghui Wu and Zhifeng Chen. GPipe: Efficient training of giant neural networks using pipeline parallelism. Advances in neural information processing systems 32 2019."},{"key":"e_1_3_2_1_14_1","unstructured":"Advanced Micro Devices Inc. AMD DirectGMA. https:\/\/github.com\/GPUOpen-LibrariesAndSDKs\/DirectGMA_P2P 2023."},{"key":"e_1_3_2_1_15_1","unstructured":"Advanced Micro Devices Inc. AMD Infinity Architecture. https:\/\/www.amd.com\/en\/technologies\/infinity-architecture 2023."},{"key":"e_1_3_2_1_16_1","unstructured":"Advanced Micro Devices Inc. MI200 high-performance computing and tuning guide. https:\/\/docs.amd.com\/en\/latest\/how-to\/tuning-guides\/mi200.html 2023."},{"key":"e_1_3_2_1_17_1","unstructured":"Advanced Micro Devices Inc. RCCL. https:\/\/github.com\/ROCmSoftwarePlatform\/rccl 2023."},{"key":"e_1_3_2_1_18_1","unstructured":"Amazon Web Services Inc. Amazon machine learning. https:\/\/docs.aws.amazon.com\/machine-learning 2023."},{"key":"e_1_3_2_1_19_1","volume-title":"Intel\u00ae QuickPath Interconnect. https:\/\/www.intel.com\/content\/www\/us\/en\/io\/quickpath-technology\/quickpath-technology-general.html","year":"2009","unstructured":"Intel. Intel\u00ae QuickPath Interconnect. https:\/\/www.intel.com\/content\/www\/us\/en\/io\/quickpath-technology\/quickpath-technology-general.html, 2009."},{"key":"e_1_3_2_1_20_1","volume-title":"oneCCL. https:\/\/github.com\/oneapi-src\/oneCCL","year":"2023","unstructured":"Intel. oneCCL. https:\/\/github.com\/oneapi-src\/oneCCL, 2023."},{"key":"e_1_3_2_1_21_1","volume-title":"Tuning Guides for Intel\u00ae Xeon\u00ae Scalable Processor-Based Systems. https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/articles\/guide\/xeon-performance-tuning-and-solution-guides.html","year":"2023","unstructured":"Intel. Tuning Guides for Intel\u00ae Xeon\u00ae Scalable Processor-Based Systems. https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/articles\/guide\/xeon-performance-tuning-and-solution-guides.html, 2023."},{"key":"e_1_3_2_1_22_1","volume-title":"Deepspeed ulysses: System optimizations for enabling training of extreme long sequence transformer models. arXiv preprint arXiv:2309.14509","author":"Jacobs Sam Ade","year":"2023","unstructured":"Sam Ade Jacobs, Masahiro Tanaka, Chengming Zhang, Minjia Zhang, Leon Song, Samyam Rajbhandari, and Yuxiong He. Deepspeed ulysses: System optimizations for enabling training of extreme long sequence transformer models. arXiv preprint arXiv:2309.14509, 2023."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507778"},{"key":"e_1_3_2_1_24_1","first-page":"947","volume-title":"2019 USENIX Annual Technical Conference (USENIX ATC 19)","author":"Jeon Myeongjae","year":"2019","unstructured":"Myeongjae Jeon, Shivaram Venkataraman, Amar Phanishayee, Junjie Qian, Wencong Xiao, and Fan Yang. Analysis of large-scale multi-tenant GPU clusters for DNN training workloads. In 2019 USENIX Annual Technical Conference (USENIX ATC 19), pages 947--960, 2019."},{"key":"e_1_3_2_1_25_1","first-page":"5","article-title":"Reducing activation recomputation in large Transformer models","author":"Korthikanti Vijay A.","year":"2023","unstructured":"Vijay A. Korthikanti, Jared Casper, Sangkug Lym, Lawrence McAfee, Michael Andersch, Mohammad Shoeybi, and Bryan Catanzaro. Reducing activation recomputation in large Transformer models. Proceedings of Machine Learning and Systems, 5, 2023.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_26_1","first-page":"741","volume-title":"18th USENIX Symposium on Networked Systems Design and Implementation (NSDI 21)","author":"Lao ChonLam","year":"2021","unstructured":"ChonLam Lao, Yanfang Le, Kshiteej Mahajan, Yixi Chen, Wenfei Wu, Aditya Akella, and Michael Swift. ATP: In-network aggregation for multi-tenant learning. In 18th USENIX Symposium on Networked Systems Design and Implementation (NSDI 21), pages 741--761, 2021."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2019.2928289"},{"key":"e_1_3_2_1_28_1","volume-title":"Sequence parallelism: Long sequence training from system perspective. arXiv preprint arXiv:2105.13120","author":"Li Shenggui","year":"2021","unstructured":"Shenggui Li, Fuzhao Xue, Chaitanya Baranwal, Yongbin Li, and Yang You. Sequence parallelism: Long sequence training from system perspective. arXiv preprint arXiv:2105.13120, 2021."},{"key":"e_1_3_2_1_29_1","first-page":"82","article-title":"Discovering and exploiting locality for accelerated distributed training on the public cloud","volume":"2","author":"Luo Liang","year":"2020","unstructured":"Liang Luo, Peter West, Jacob Nelson, Arvind Krishnamurthy, and Luis Ceze. PLink: Discovering and exploiting locality for accelerated distributed training on the public cloud. Proceedings of Machine Learning and Systems, 2:82--97, 2020.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_30_1","first-page":"809","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Mahajan Kshiteej","year":"2023","unstructured":"Kshiteej Mahajan, Ching-Hsiang Chu, Srinivas Sridharan, and Aditya Akella. Better Together: Jointly optimizing ML collective scheduling and execution planning using SYNDICATE. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23), pages 809--824, 2023."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2016.62"},{"key":"e_1_3_2_1_32_1","volume-title":"https:\/\/azure.microsoft.com\/solutions\/ai","author":"Azure","year":"2023","unstructured":"Microsoft. Azure AI. https:\/\/azure.microsoft.com\/solutions\/ai, 2023."},{"key":"e_1_3_2_1_33_1","volume-title":"Zaid Qureshi, Jinjun Xiong, Eiman Ebrahimi, and Wen-mei Hwu. Emogi: Efficient memory-access for out-of-memory graph-traversal in gpus. arXiv preprint arXiv:2006.06890","author":"Min Seung Won","year":"2020","unstructured":"Seung Won Min, Vikram Sharma Mailthody, Zaid Qureshi, Jinjun Xiong, Eiman Ebrahimi, and Wen-mei Hwu. Emogi: Efficient memory-access for out-of-memory graph-traversal in gpus. arXiv preprint arXiv:2006.06890, 2020."},{"key":"e_1_3_2_1_34_1","volume-title":"Intel\u00ae Xeon\u00ae Processor Scalable Family Technical Overview. https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/articles\/technical\/xeon-processor-scalable-family-technical-overview.html","author":"Mulnix David L.","year":"2022","unstructured":"David L. Mulnix. Intel\u00ae Xeon\u00ae Processor Scalable Family Technical Overview. https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/articles\/technical\/xeon-processor-scalable-family-technical-overview.html, 2022."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476209"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3230543.3230560"},{"key":"e_1_3_2_1_37_1","volume-title":"https:\/\/developer.download.nvidia.com\/CUDA\/training\/cuda_webinars_GPUDirect_uva.pdf","author":"Peer NVIDIA.","year":"2011","unstructured":"NVIDIA. Peer-to-Peer & Unified Virtual Addressing. https:\/\/developer.download.nvidia.com\/CUDA\/training\/cuda_webinars_GPUDirect_uva.pdf, 2011."},{"key":"e_1_3_2_1_38_1","volume-title":"NVIDIA NVSWITCH - The World's Highest-Bandwidth On-Node Switch. https:\/\/images.nvidia.com\/content\/pdf\/nvswitch-technical-overview.pdf","author":"NVIDIA.","year":"2018","unstructured":"NVIDIA. NVIDIA NVSWITCH - The World's Highest-Bandwidth On-Node Switch. https:\/\/images.nvidia.com\/content\/pdf\/nvswitch-technical-overview.pdf, 2018."},{"key":"e_1_3_2_1_39_1","volume-title":"cuBLAS. https:\/\/developer.nvidia.com\/cublas","author":"NVIDIA.","year":"2023","unstructured":"NVIDIA. cuBLAS. https:\/\/developer.nvidia.com\/cublas, 2023."},{"key":"e_1_3_2_1_40_1","volume-title":"https:\/\/github.com\/NVIDIA\/nccl","author":"NVIDIA.","year":"2023","unstructured":"NVIDIA. NCCL. https:\/\/github.com\/NVIDIA\/nccl, 2023."},{"key":"e_1_3_2_1_41_1","volume-title":"nvbandwidth. https:\/\/github.com\/NVIDIA\/nvbandwidth","author":"NVIDIA.","year":"2023","unstructured":"NVIDIA. nvbandwidth. https:\/\/github.com\/NVIDIA\/nvbandwidth, 2023."},{"key":"e_1_3_2_1_42_1","volume-title":"NVIDIA cuDNN. https:\/\/developer.nvidia.com\/cudnn","author":"NVIDIA.","year":"2023","unstructured":"NVIDIA. NVIDIA cuDNN. https:\/\/developer.nvidia.com\/cudnn, 2023."},{"key":"e_1_3_2_1_43_1","volume-title":"https:\/\/developer.nvidia.com\/gpudirect","author":"Direct NVIDIA. NVIDIA","year":"2023","unstructured":"NVIDIA. NVIDIA GPUDirect. https:\/\/developer.nvidia.com\/gpudirect, 2023."},{"key":"e_1_3_2_1_44_1","volume-title":"https:\/\/www.nvidia.com\/en-us\/design-visualization\/nvlink-bridges\/","author":"Link NVIDIA.","year":"2023","unstructured":"NVIDIA. NVLink. https:\/\/www.nvidia.com\/en-us\/design-visualization\/nvlink-bridges\/, 2023."},{"key":"e_1_3_2_1_45_1","first-page":"8024","volume-title":"Proceedings of the 33rd International Conference on Neural Information Processing Systems","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, Alban Desmaison, Andreas K\u00f6pf, Edward Yang, Zachary DeVito, Martin Raison, Alykhan Tejani, Sasank Chilamkurthy, Benoit Steiner, Lu Fang, Junjie Bai, and Soumith Chintala. Pytorch: An imperative style, high-performance deep learning library. In Proceedings of the 33rd International Conference on Neural Information Processing Systems, pages 8024--8035. 2019."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3297663.3310299"},{"key":"e_1_3_2_1_47_1","volume-title":"Towards automatic and adaptive optimizations of MPI collective operations","author":"Pjesivac-Grbovic Jelena","year":"2007","unstructured":"Jelena Pjesivac-Grbovic. Towards automatic and adaptive optimizations of MPI collective operations. 2007."},{"key":"e_1_3_2_1_48_1","volume-title":"GPU-initiated fine-grained overlap of collective communication with computation. arXiv preprint arXiv:2305.06942","author":"Punniyamurthy Kishore","year":"2023","unstructured":"Kishore Punniyamurthy, Bradford M. Beckmann, and Khaled Hamidouche. GPU-initiated fine-grained overlap of collective communication with computation. arXiv preprint arXiv:2305.06942, 2023."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-30218-6_13"},{"issue":"8","key":"e_1_3_2_1_50_1","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford Alec","year":"2019","unstructured":"Alec Radford, Jeffrey Wu, Rewon Child, David Luan, Dario Amodei, and Ilya Sutskever. Language models are unsupervised multitask learners. OpenAI blog, 1(8):9, 2019.","journal-title":"OpenAI blog"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.5555\/3455716.3455856"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00049"},{"key":"e_1_3_2_1_54_1","volume-title":"AMD Optimizes EPYC Memory with NUMA. https:\/\/www.amd.com\/content\/dam\/amd\/en\/documents\/epyc-business-docs\/white-papers\/AMD-Optimizes-EPYC-Memory-With-NUMA.pdf","author":"Research TIRIAS","year":"2018","unstructured":"TIRIAS Research. AMD Optimizes EPYC Memory with NUMA. https:\/\/www.amd.com\/content\/dam\/amd\/en\/documents\/epyc-business-docs\/white-papers\/AMD-Optimizes-EPYC-Memory-With-NUMA.pdf, 2018."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.parco.2009.09.001"},{"key":"e_1_3_2_1_56_1","first-page":"56","volume-title":"2023 IEEE International Symposium on High-Performance Computer Architecture (HPCA)","author":"Sanghoon Jo","year":"2023","unstructured":"Jo Sanghoon, Hyojun Son, and John Kim. Logical\/physical topology-aware collective communication in deep learning training. In 2023 IEEE International Symposium on High-Performance Computer Architecture (HPCA), pages 56--68. IEEE, 2023."},{"key":"e_1_3_2_1_57_1","first-page":"785","volume-title":"18th USENIX Symposium on Networked Systems Design and Implementation (NSDI 21)","author":"Sapio Amedeo","year":"2021","unstructured":"Amedeo Sapio, Marco Canini, Chen-Yu Ho, Jacob Nelson, Panos Kalnis, Changhoon Kim, Arvind Krishnamurthy, Masoud Moshref, Dan Ports, and Peter Richt\u00e1rik. Scaling distributed machine learning with in-network aggregation. In 18th USENIX Symposium on Networked Systems Design and Implementation (NSDI 21), pages 785--808, 2021."},{"key":"e_1_3_2_1_58_1","first-page":"593","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Shah Aashaka","year":"2023","unstructured":"Aashaka Shah, Vijay Chidambaram, Meghan Cowan, Saeed Maleki, Madan Musuvathi, Todd Mytkowicz, Jacob Nelson, Olli Saarikivi, and Rachee Singh. TACCL: Guiding collective algorithm synthesis using communication sketches. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23), pages 593--612, 2023."},{"key":"e_1_3_2_1_59_1","volume-title":"Megatron-LM: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053","author":"Shoeybi Mohammad","year":"2019","unstructured":"Mohammad Shoeybi, Mostofa Patwary, Raul Puri, Patrick LeGresley, Jared Casper, and Bryan Catanzaro. Megatron-LM: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053, 2019."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342005051521"},{"key":"e_1_3_2_1_61_1","first-page":"172","article-title":"Fast and generic collectives for distributed ML","volume":"2","author":"Wang Guanhua","year":"2020","unstructured":"Guanhua Wang, Shivaram Venkataraman, Amar Phanishayee, Nikhil Devanur, Jorgen Thelin, and Ion Stoica. Blink: Fast and generic collectives for distributed ML. Proceedings of Machine Learning and Systems, 2:172--186, 2020.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3567955.3567959"},{"key":"e_1_3_2_1_63_1","first-page":"945","volume-title":"19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22)","author":"Weng Qizhen","year":"2022","unstructured":"Qizhen Weng, Wencong Xiao, Yinghao Yu, Wei Wang, Cheng Wang, Jian He, Yong Li, Liping Zhang, Wei Lin, and Yu Ding. MLaaS in the wild: Workload analysis and scheduling in large-scale heterogeneous GPU clusters. In 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22), pages 945--960, 2022."},{"key":"e_1_3_2_1_64_1","volume-title":"TACOS: Topology-aware collective algorithm synthesizer for distributed training. arXiv preprint arXiv:2304.05301","author":"Won William","year":"2023","unstructured":"William Won, Midhilesh Elavazhagan, Sudarshan Srinivasan, Ajaya Durg, Swati Gupta, and Tushar Krishna. TACOS: Topology-aware collective algorithm synthesizer for distributed training. arXiv preprint arXiv:2304.05301, 2023."},{"key":"e_1_3_2_1_65_1","first-page":"548","article-title":"Synthesizing optimal parallelism placement and reduction strategies on hierarchical systems for deep learning","volume":"4","author":"Xie Ningning","year":"2022","unstructured":"Ningning Xie, Tamara Norman, Dominik Grewe, and Dimitrios Vytiniotis. Synthesizing optimal parallelism placement and reduction strategies on hierarchical systems for deep learning. Proceedings of Machine Learning and Systems, 4:548--566, 2022. Received 30 November 2023; accepted 6 March 2024","journal-title":"Proceedings of Machine Learning and Systems"}],"event":{"name":"ASPLOS '24: 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 3","location":"La Jolla CA USA","acronym":"ASPLOS '24","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture","SIGOPS ACM Special Interest Group on Operating Systems","SIGPLAN ACM Special Interest Group on Programming Languages","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 3"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3620666.3651362","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T20:03:43Z","timestamp":1750277023000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3620666.3651362"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,27]]},"references-count":65,"alternative-id":["10.1145\/3620666.3651362","10.1145\/3620666"],"URL":"https:\/\/doi.org\/10.1145\/3620666.3651362","relation":{},"subject":[],"published":{"date-parts":[[2024,4,27]]},"assertion":[{"value":"2024-04-27","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}