{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T05:12:12Z","timestamp":1783746732747,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","funder":[{"name":"Science and Education Research Board (SERB)","award":["SRG\\\/2023\\\/002445"],"award-info":[{"award-number":["SRG\\\/2023\\\/002445"]}]},{"name":"BITS ACG","award":["GOA\\\/ACG\\\/2022-2023\\\/Oct\\\/11"],"award-info":[{"award-number":["GOA\\\/ACG\\\/2022-2023\\\/Oct\\\/11"]}]},{"name":"BITS CDRF","award":["C1\\\/23\\\/173"],"award-info":[{"award-number":["C1\\\/23\\\/173"]}]},{"name":"LLNL LDRD","award":["25-ERD-042"],"award-info":[{"award-number":["25-ERD-042"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,13]]},"DOI":"10.1145\/3806645.3816128","type":"proceedings-article","created":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T04:21:11Z","timestamp":1783743671000},"page":"807-816","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["CoDL: A Framework for Studying Cross-Component Interference in Deep Learning Training Pipelines"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-5466-8473","authenticated-orcid":false,"given":"Druva","family":"Dhakshinamoorthy","sequence":"first","affiliation":[{"name":"BITS Pilani, KK Birla Goa Campus, Sancaole, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7255-7647","authenticated-orcid":false,"given":"Ray A. O.","family":"Sinurat","sequence":"additional","affiliation":[{"name":"The University of Chicago, Chicago, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9965-3647","authenticated-orcid":false,"given":"Nikoli","family":"Dryden","sequence":"additional","affiliation":[{"name":"Lawrence Livermore National Laboratory, Livermore, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3694-5511","authenticated-orcid":false,"given":"Arnab K.","family":"Paul","sequence":"additional","affiliation":[{"name":"BITS Pilani, KK Birla Goa Campus, Sancaole, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5625-3494","authenticated-orcid":false,"given":"Hariharan","family":"Devarajan","sequence":"additional","affiliation":[{"name":"Lawrence Livermore National Laboratory, Livermore, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,13]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"crossref","unstructured":"L. Adhianto S. Banerjee M. Fagan M. Krentel G. Marin J. Mellor-Crummey and N.\u00a0R. Tallent. 2010. HPCTOOLKIT: tools for performance analysis of optimized parallel programs http:\/\/hpctoolkit.org. Concurr. Comput. : Pract. Exper. 22 6 (April 2010) 685\u2013701.","DOI":"10.1002\/cpe.1553"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640366"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620678.3624666"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-99854-6_4"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","unstructured":"Tal Ben-Nun and Torsten Hoefler. 2019. Demystifying Parallel and Distributed Deep Learning: An In-Depth Concurrency Analysis. Comput. Surveys 52 4 (2019) 1\u201343. 10.1145\/3320060","DOI":"10.1145\/3320060"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","unstructured":"D Boehme T Gamblin D Beckingsale P Bremer A Gimenez M LeGendre O Pearce and M Schulz. 2016. Caliper: Performance Introspection for HPC Software Stacks. Lawrence Livermore National Laboratory (LLNL) Livermore CA (United States). 10.1109\/SC.2016.46","DOI":"10.1109\/SC.2016.46"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTR.2009.5289150"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607060"},{"key":"e_1_3_3_1_10_2","volume-title":"NeurIPS ML Systems Workshop","author":"Coleman Cody","year":"2017","unstructured":"Cody Coleman, Deepak Narayanan, Daniel Kang, Tian Zhao, Jian Zhang, Luigi Nardi, Peter Bailis, Kunle Olukotun, Christopher R\u00e9, and Matei Zaharia. 2017. DAWNBench: An End-to-End Deep Learning Benchmark and Competition. In NeurIPS ML Systems Workshop."},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41406.2024.00023"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid51090.2021.00018"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476181"},{"key":"e_1_3_3_1_14_2","volume-title":"Workshop on Machine Learning in High Performance Computing Environments (MLHPC)","author":"Farrell Steven","year":"2021","unstructured":"Steven Farrell, Murali Emani, Jacob Balma, Lukas Drescher, Aleksandr Drozd, Andreas Fink, Geoffrey Fox, David Kanter, Thorsten Kurth, Peter Mattson, et\u00a0al. 2021. MLPerf HPC: A Holistic Benchmark Suite for Scientific Machine Learning on HPC Systems. In Workshop on Machine Learning in High Performance Computing Environments (MLHPC). IEEE."},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","unstructured":"Shaoduo Gan Jiawei Jiang Binhang Yuan Ce Zhang Xiangru Lian Rui Wang Jianbin Chang Chengjun Liu Hongmei Shi Shengzhuo Zhang Xianghong Li Tengxu Sun Sen Yang and Ji Liu. 2021. Bagua: scaling up distributed learning with system relaxations. Proc. VLDB Endow. 15 4 (Dec. 2021) 804\u2013813. 10.14778\/3503585.3503590","DOI":"10.14778\/3503585.3503590"},{"key":"e_1_3_3_1_16_2","volume-title":"Proceedings of the USENIX Symposium on Operating Systems Design and Implementation (OSDI)","author":"Gujarati Arpan","year":"2020","unstructured":"Arpan Gujarati, Reza Karimi, Safya Alzayat, Wei Hao, Antoine Kaufmann, Ymir Vigfusson, and Jonathan Mace. 2020. Serving DNNs like Clockwork: Performance Predictability from the Bottom Up. In Proceedings of the USENIX Symposium on Operating Systems Design and Implementation (OSDI)."},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476223"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.1145\/3514221.3517848"},{"key":"e_1_3_3_1_19_2","volume-title":"GPU Technology Conference (GTC)","author":"Jeaugey Sylvain","year":"2017","unstructured":"Sylvain Jeaugey. 2017. NCCL 2.0: Optimized Primitives for Collective Multi-GPU Communication. In GPU Technology Conference (GTC). NVIDIA. https:\/\/developer.nvidia.com\/nccl."},{"key":"e_1_3_3_1_20_2","first-page":"947","volume-title":"Proceedings of USENIX Annual Technical Conference (ATC)","author":"Jeon Myeongjae","year":"2019","unstructured":"Myeongjae Jeon, Shivaram Venkataraman, Amar Phanishayee, Junjie Qian, Wencong Xiao, and Fan Yang. 2019. Analysis of Large-Scale Multi-Tenant GPU Clusters for DNN Training Workloads. In Proceedings of USENIX Annual Technical Conference (ATC). 947\u2013960."},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","unstructured":"Danlin Jia Geng Yuan Yiming Xie Xue Lin and Ningfang Mi. 2024. A Data-Loader Tunable Knob to Shorten GPU Idleness for Distributed Deep Learning. ACM Transactions on Architecture and Code Optimization (TACO) 21 4 (2024). 10.1145\/3680546","DOI":"10.1145\/3680546"},{"key":"e_1_3_3_1_22_2","series-title":"(OSDI\u201920)","volume-title":"Proceedings of the 14th USENIX Conference on Operating Systems Design and Implementation","author":"Jiang Yimin","year":"2020","unstructured":"Yimin Jiang, Yibo Zhu, Chang Lan, Bairen Yi, Yong Cui, and Chuanxiong Guo. 2020. A unified architecture for accelerating distributed DNN training in heterogeneous GPU\/CPU clusters. In Proceedings of the 14th USENIX Conference on Operating Systems Design and Implementation(OSDI\u201920). USENIX Association, USA, Article 26, 17\u00a0pages."},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","unstructured":"Andreas Kn\u00fcpfer Christian Feld Dieter Mey Scott Biersdorff Kai Diethelm Dominic Eschweiler Markus Geimer Michael Gerndt Daniel Lorenz Allen Malony Wolfgang Nagel Yury Oleynik Peter Philippen Pavel Saviankou Dirk Schmidl Sameer Shende Ronny Tsch\u00fcter Michael Wagner Bert Wesarg and Felix Wolf. 2012. Score-P: A Joint Performance Measurement Run-Time Infrastructure for Periscope Scalasca TAU and Vampir. 79\u201391. 10.1007\/978-3-642-31476-6_7","DOI":"10.1007\/978-3-642-31476-6_7"},{"key":"e_1_3_3_1_24_2","volume-title":"Proceedings of Machine Learning and Systems (MLSys)","author":"Kuchnik Michael","year":"2022","unstructured":"Michael Kuchnik, Ana Klimovic, Jiri Simsa, Virginia Smith, and George Amvrosiadis. 2022. Plumber: Diagnosing and Removing Performance Bottlenecks in Machine Learning Data Pipelines. In Proceedings of Machine Learning and Systems (MLSys)."},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00054"},{"key":"e_1_3_3_1_26_2","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Leclerc Guillaume","year":"2023","unstructured":"Guillaume Leclerc, Andrew Ilyas, Logan Engstrom, Sung\u00a0Min Park, Hadi Salman, and Aleksander Madry. 2023. FFCV: Accelerating Training by Removing Data Bottlenecks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS47924.2020.00042"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","unstructured":"Shen Li Yanli Zhao Rohan Varma Omkar Salpekar Pieter Noordhuis Teng Li Adam Paszke Jeff Smith Brian Vaughan Pritam Damania and Soumith Chintala. 2020. PyTorch distributed: experiences on accelerating data parallel training. Proc. VLDB Endow. 13 12 3005\u20133018. 10.14778\/3415478.3415530","DOI":"10.14778\/3415478.3415530"},{"key":"e_1_3_3_1_29_2","first-page":"336","volume-title":"Proceedings of Machine Learning and Systems","volume":"2","author":"Mattson Peter","year":"2020","unstructured":"Peter Mattson, Christine Cheng, Gregory Diamos, Cody Coleman, Paulius Micikevicius, David Patterson, Hanlin Tang, Gu-Yeon Wei, Peter Bailis, Victor Bittorf, David Brooks, Dehao Chen, Debo Dutta, Udit Gupta, Kim Hazelwood, Andy Hock, Xinyuan Huang, Daniel Kang, David Kanter, Naveen Kumar, Jeffery Liao, Deepak Narayanan, Tayo Oguntebi, Gennady Pekhimenko, Lillian Pentecost, Vijay Janapa\u00a0Reddi, Taylor Robie, Tom St\u00a0John, Carole-Jean Wu, Lingjie Xu, Cliff Young, and Matei Zaharia. 2020. MLPerf Training Benchmark. In Proceedings of Machine Learning and Systems , I.\u00a0Dhillon, D.\u00a0Papailiopoulos, and V.\u00a0Sze (Eds.), Vol.\u00a02. 336\u2013349. https:\/\/proceedings.mlsys.org\/paper_files\/paper\/2020\/file\/411e39b117e885341f25efb8912945f7-Paper.pdf"},{"key":"e_1_3_3_1_30_2","unstructured":"Microsoft Azure HPC Team. 2024. Optimizing AI Workloads on Azure: CPU Pinning via NCCL Topology File. https:\/\/techcommunity.microsoft.com\/blog\/azurehighperformancecomputingblog\/optimizing-ai-workloads-on-azure-cpu-pinning-via-nccl-topology-file\/4371810."},{"key":"e_1_3_3_1_31_2","unstructured":"MLCommons. 2024. MLPerf Storage Benchmark Suite. https:\/\/mlcommons.org\/benchmarks\/storage\/. Built on top of DLIO as the workload generator."},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"publisher","unstructured":"Jayashree Mohan Amar Phanishayee Ashish Raniwala and Vijay Chidambaram. 2021. Analyzing and Mitigating Data Stalls in DNN Training. Proceedings of the VLDB Endowment 14 5 (2021) 771\u2013784. 10.14778\/3446095.3446100","DOI":"10.14778\/3446095.3446100"},{"key":"e_1_3_3_1_33_2","doi-asserted-by":"publisher","unstructured":"Derek\u00a0G. Murray Jiri Simsa Ana Klimovic and Ihor Indyk. 2021. tf.data: A Machine Learning Data Processing Framework. Proceedings of the VLDB Endowment 14 12 (2021) 2945\u20132958. 10.14778\/3476311.3476374","DOI":"10.14778\/3476311.3476374"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3369583.3392674"},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"publisher","DOI":"10.1145\/3767295.3769376"},{"key":"e_1_3_3_1_37_2","unstructured":"NVIDIA Corporation. 2024. CUDA Multi-Process Service. https:\/\/docs.nvidia.com\/deploy\/mps\/index.html."},{"key":"e_1_3_3_1_38_2","unstructured":"NVIDIA Corporation. 2024. CUPTI: CUDA Profiling Tools Interface. https:\/\/developer.nvidia.com\/cupti."},{"key":"e_1_3_3_1_39_2","unstructured":"NVIDIA Corporation. 2024. NVIDIA DALI: Data Loading Library. https:\/\/developer.nvidia.com\/dali."},{"key":"e_1_3_3_1_40_2","unstructured":"NVIDIA Corporation. 2024. NVIDIA Nsight Systems. https:\/\/developer.nvidia.com\/nsight-systems."},{"key":"e_1_3_3_1_41_2","unstructured":"PyTorch Contributors. 2023. PyTorch Profiler. https:\/\/pytorch.org\/tutorials\/recipes\/recipes\/profiler_recipe.html."},{"key":"e_1_3_3_1_42_2","doi-asserted-by":"publisher","unstructured":"Sameer\u00a0S. Shende and Allen\u00a0D. Malony. 2006. The Tau Parallel Performance System. The International Journal of High Performance Computing Applications 20 2 (2006) 287\u2013311. arXiv:https:\/\/doi.org\/10.1177\/109434200606448210.1177\/1094342006064482","DOI":"10.1177\/1094342006064482"},{"key":"e_1_3_3_1_43_2","doi-asserted-by":"publisher","unstructured":"Fei Tang Wanling Gao Jianfeng Zhan et\u00a0al. 2020. AIBench Training: Balanced Industry-Standard AI Training Benchmarking. (2020). arxiv:https:\/\/arXiv.org\/abs\/2004.1469010.48550\/arXiv.2004.14690","DOI":"10.48550\/arXiv.2004.14690"},{"key":"e_1_3_3_1_44_2","doi-asserted-by":"publisher","unstructured":"Shu-Mei Tseng Bogdan Nicolae Franck Cappello and Aparna Chandramowlishwaran. 2021. Demystifying asynchronous I\/O Interference in HPC applications. Int. J. High Perform. Comput. Appl. 35 4 (July 2021) 391\u2013412. 10.1177\/10943420211016511","DOI":"10.1177\/10943420211016511"},{"key":"e_1_3_3_1_45_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW50202.2020.00176"},{"key":"e_1_3_3_1_46_2","doi-asserted-by":"publisher","DOI":"10.1145\/3470496.3533044"},{"key":"e_1_3_3_1_47_2","first-page":"88","volume-title":"Proceedings of the IEEE International Symposium on Workload Characterization (IISWC)","author":"Zhu Hongyu","year":"2018","unstructured":"Hongyu Zhu, Mohamed Akrout, Bojian Zheng, Andrew Pelegris, Anand Jayarajan, Amar Phanishayee, Bianca Schroeder, and Gennady Pekhimenko. 2018. Benchmarking and Analyzing Deep Neural Network Training. In Proceedings of the IEEE International Symposium on Workload Characterization (IISWC). IEEE, 88\u2013100."}],"event":{"name":"HPDC '26: 35th International Symposium on High-Performance Parallel and Distributed Computing","location":"Cleveland USA","acronym":"HPDC '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 35th International Symposium on High-Performance Parallel and Distributed Computing"],"original-title":[],"deposited":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T04:21:46Z","timestamp":1783743706000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3806645.3816128"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,13]]},"references-count":46,"alternative-id":["10.1145\/3806645.3816128","10.1145\/3806645"],"URL":"https:\/\/doi.org\/10.1145\/3806645.3816128","relation":{},"subject":[],"published":{"date-parts":[[2026,7,13]]},"assertion":[{"value":"2026-07-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}