{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,2]],"date-time":"2025-08-02T15:02:53Z","timestamp":1754146973695,"version":"3.41.2"},"publisher-location":"New York, NY, USA","reference-count":13,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,7,20]]},"DOI":"10.1145\/3708035.3736055","type":"proceedings-article","created":{"date-parts":[[2025,7,18]],"date-time":"2025-07-18T12:10:41Z","timestamp":1752840641000},"page":"1-5","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["InterconnectLens: Enhancing Observability of Data Transfers in GPU Clusters"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-5343-9073","authenticated-orcid":false,"given":"Koshi","family":"Eguchi","sequence":"first","affiliation":[{"name":"The University of Tokyo, Bunkyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6664-7534","authenticated-orcid":false,"given":"Ryo","family":"Nakamura","sequence":"additional","affiliation":[{"name":"The University of Tokyo, Bunkyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8586-0533","authenticated-orcid":false,"given":"Yohei","family":"Kuga","sequence":"additional","affiliation":[{"name":"Toyota Motor Corporation, Otemachi, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5224-382X","authenticated-orcid":false,"given":"Kenjiro","family":"Taura","sequence":"additional","affiliation":[{"name":"The University of Tokyo, Bunkyo, Japan and NII LLMC, Chiyoda, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,7,18]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Ege Erdil and David Schneider-Joseph. 2024. Data movement limits to frontier model training. arxiv:https:\/\/arXiv.org\/abs\/2411.01137\u00a0[cs.DC] https:\/\/arxiv.org\/abs\/2411.01137"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","unstructured":"Amir Gholami Zhewei Yao Sehoon Kim Coleman Hooper Michael\u00a0W. Mahoney and Kurt Keutzer. 2024. AI and Memory Wall. IEEE Micro 44 3 (May 2024) 33\u201339. 10.1109\/MM.2024.3373763","DOI":"10.1109\/MM.2024.3373763"},{"key":"e_1_3_3_1_4_2","unstructured":"Marius Hobbhahn Lennart Heim and G\u00f6k\u00e7e Aydos. 2023. Trends in Machine Learning Hardware. https:\/\/epoch.ai\/blog\/trends-in-machine-learning-hardware Accessed: 2025-06-19."},{"key":"e_1_3_3_1_5_2","unstructured":"Intel. 2025. intel\/pcm: Intel\u00ae Performance Counter Monitor (Intel\u00ae PCM). https:\/\/github.com\/intel\/pcm. Accessed: 2025-06-19."},{"key":"e_1_3_3_1_6_2","unstructured":"Jared Kaplan Sam McCandlish Tom Henighan Tom\u00a0B. Brown Benjamin Chess Rewon Child Scott Gray Alec Radford Jeffrey Wu and Dario Amodei. 2020. Scaling Laws for Neural Language Models. arxiv:https:\/\/arXiv.org\/abs\/2001.08361\u00a0[cs.LG]"},{"key":"e_1_3_3_1_7_2","unstructured":"NVIDIA. 2025. Developing a linux kernel module using RDMA for GPUdirect. https:\/\/docs.nvidia.com\/cuda\/gpudirect-rdma\/ Accessed: 2025-06-19."},{"key":"e_1_3_3_1_8_2","unstructured":"NVIDIA. 2025. Nsight Systems | NVIDIA Developer. https:\/\/docs.nvidia.com\/nsight-systems\/index.html Accessed: 2025-06-19."},{"key":"e_1_3_3_1_9_2","unstructured":"NVIDIA. 2025. NVIDIA\/dcgm-exporter: NVIDIA GPU metrics exporter. https:\/\/github.com\/NVIDIA\/dcgm-exporter. Accessed: 2025-06-19."},{"key":"e_1_3_3_1_10_2","unstructured":"NVIDIA. 2025. NVIDIA\/nccl: Optimized primitives for collective multi-GPU communication. https:\/\/github.com\/nvidia\/nccl. Accessed: 2025-06-19."},{"key":"e_1_3_3_1_11_2","unstructured":"Prometheus. 2025. Prometheus - Monitoring system & time series database. https:\/\/prometheus.io\/docs\/introduction\/overview\/. Accessed: 2025-06-19."},{"key":"e_1_3_3_1_12_2","unstructured":"Prometheus. 2025. prometheus\/procfs: procfs provides functions to retrieve system kernel and process metrics from the pseudo-filesystem proc.https:\/\/github.com\/prometheus\/procfs Accessed: 2025-06-19."},{"key":"e_1_3_3_1_13_2","unstructured":"Prometheus. 2025. Querying basics | Prometheus. https:\/\/prometheus.io\/docs\/prometheus\/latest\/querying\/basics\/ [Online; accessed 2025-06-19]."},{"key":"e_1_3_3_1_14_2","unstructured":"Jaime Sevilla and Edu Rold\u00e1n. 2024. Training Compute of Frontier AI Models Grows by 4-5x per Year. https:\/\/epoch.ai\/blog\/training-compute-of-frontier-ai-models-grows-by-4-5x-per-year Accessed: 2025-06-19."}],"event":{"name":"PEARC '25: Practice and Experience in Advanced Research Computing","location":"Columbus Ohio USA","acronym":"PEARC '25","sponsor":["SIGAPP ACM Special Interest Group on Applied Computing","SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Practice and Experience in Advanced Research Computing 2025: The Power of Collaboration"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3708035.3736055","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,18]],"date-time":"2025-07-18T12:32:45Z","timestamp":1752841965000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3708035.3736055"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,18]]},"references-count":13,"alternative-id":["10.1145\/3708035.3736055","10.1145\/3708035"],"URL":"https:\/\/doi.org\/10.1145\/3708035.3736055","relation":{},"subject":[],"published":{"date-parts":[[2025,7,18]]},"assertion":[{"value":"2025-07-18","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}