{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T15:33:48Z","timestamp":1772724828606,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":41,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,12]],"date-time":"2024-08-12T00:00:00Z","timestamp":1723420800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Natural Science Foundation of China Youth Science Foundation Project","award":["62102438"],"award-info":[{"award-number":["62102438"]}]},{"name":"National Natural Science Foundation of China Youth Science Foundation Project","award":["62102439"],"award-info":[{"award-number":["62102439"]}]},{"name":"Beijing Nova Program","award":["20220484150"],"award-info":[{"award-number":["20220484150"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,12]]},"DOI":"10.1145\/3673038.3673075","type":"proceedings-article","created":{"date-parts":[[2024,8,8]],"date-time":"2024-08-08T18:29:01Z","timestamp":1723141741000},"page":"1001-1011","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["AdCoalescer: An Adaptive Coalescer to Reduce the Inter-Module Traffic in MCM-GPUs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-8856-8278","authenticated-orcid":false,"given":"Xu","family":"Zhang","sequence":"first","affiliation":[{"name":"Defense Innovation Institute, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4732-9674","authenticated-orcid":false,"given":"Guangda","family":"Zhang","sequence":"additional","affiliation":[{"name":"Defense Innovation Institute, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5759-6544","authenticated-orcid":false,"given":"Lu","family":"Wang","sequence":"additional","affiliation":[{"name":"Defense Innovation Institute, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6690-3718","authenticated-orcid":false,"given":"Shiqing","family":"Zhang","sequence":"additional","affiliation":[{"name":"Defense Innovation Institute, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6479-9200","authenticated-orcid":false,"given":"Xia","family":"Zhao","sequence":"additional","affiliation":[{"name":"Defense Innovation Institute, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,8,12]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2022. NVIDIA NVLink High-Speed GPU Interconnect. https:\/\/www.nvidia.com\/en-us\/design-visualization\/nvlink-bridges\/. NVIDIA."},{"key":"e_1_3_2_1_2_1","volume-title":"Wilson Wai\u00a0Lun Fung, and Timothy\u00a0G. Rogers","author":"Aamodt M.","year":"2018","unstructured":"Tor\u00a0M. Aamodt, Wilson Wai\u00a0Lun Fung, and Timothy\u00a0G. Rogers. 2018. General-Purpose Graphics Processor Architectures. Morgan & Claypool Publishers."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080231"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2019.00063"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2009.4919648"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00055"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2017.58"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"e_1_3_2_1_9_1","volume-title":"Intel: Sapphire Rapids with 64GB of HBM2e, Ponte Vecchio with 408 MB L2 Cache. https:\/\/www.anandtech.com\/show\/17067\/intel-sapphire-rapids-with-64-gb-of-hbm2e-ponte-vecchio-with-408-mb-l2-cache.","author":"Cutress Ian","year":"2021","unstructured":"Ian Cutress. 2021. Intel: Sapphire Rapids with 64GB of HBM2e, Ponte Vecchio with 408 MB L2 Cache. https:\/\/www.anandtech.com\/show\/17067\/intel-sapphire-rapids-with-64-gb-of-hbm2e-ponte-vecchio-with-408-mb-l2-cache."},{"key":"e_1_3_2_1_10_1","unstructured":"J. Dean G. Corrado R. Monga and et al.2012. Large Scale Distributed Deep Networks. In Advances in Neural Information Processing Systems Vol.\u00a025."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10070981"},{"key":"e_1_3_2_1_12_1","unstructured":"I. Goodfellow Y. Bengio and A. Courville. 2016. Deep Learning. MIT Press."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2014.61"},{"key":"e_1_3_2_1_14_1","volume-title":"On-Chip Networks","author":"Jerger Natalie\u00a0Enright","unstructured":"Natalie\u00a0Enright Jerger, Tushar Krishna, and Li-Shiuan Peh. 2017. On-Chip Networks: Second Edition (2nd ed.). Morgan & Claypool Publishers.","edition":"2"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3313231.3352368"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2016.53"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO50266.2020.00086"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3079079.3079088"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2015.2414456"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2015.2435709"},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the International Symposium on Computer Architecture (ISCA). 166\u2013179","author":"Liu Y.","unstructured":"Y. Liu, X. Zhao, M. Jahre, Z. Wang, X. Wang, Y. Luo, and L. Eeckhout. 2018. Get Out of the Valley: Power-Efficient Address Mapping for GPUs. In Proceedings of the International Symposium on Computer Architecture (ISCA). 166\u2013179."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123939.3124534"},{"key":"e_1_3_2_1_23_1","volume-title":"A Tool to Model Large Caches. HP laboratories 27","author":"Muralimanohar Naveen","year":"2009","unstructured":"Naveen Muralimanohar, Rajeev Balasubramonian, and Norman\u00a0P Jouppi. 2009. CACTI 6.0: A Tool to Model Large Caches. HP laboratories 27 (2009), 28."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/1401132.1401152"},{"key":"e_1_3_2_1_25_1","unstructured":"Nvidia. 2024. NVIDIA CUDA SDK Code Samples. https:\/\/developer.nvidia.com\/cuda-downloads."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2019.00042"},{"key":"e_1_3_2_1_27_1","volume-title":"Solid-State Circuits","author":"Poulton W.","year":"2013","unstructured":"John\u00a0W. Poulton, William\u00a0J. Dally, Xi Chen, John\u00a0G. Eyles, Thomas\u00a0H. Greer, Stephen\u00a0G. Tell, John\u00a0M. Wilson, and C.\u00a0Thomas Gray. 2013. A 0.54 pJ\/b 20 Gb\/s Ground-Referenced Single-Ended Short-Reach Serial Link in 28 nm CMOS for Advanced Packaging Applications. Solid-State Circuits, IEEE Journal of12 (2013), 3206\u20133218."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO56248.2022.00036"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2012.16"},{"key":"e_1_3_2_1_30_1","unstructured":"Ryan Smith. 2021. AMD Announces Instinct MI200 Accelerator Family."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2013.73"},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the International Symposium on Networks-on-Chip (NOCS). 201\u2013210","author":"Sun C.","unstructured":"C. Sun, C.\u00a0H.\u00a0O. Chen, G. Kurian, L. Wei, J. Miller, A. Agarwal, L.\u00a0S. Peh, and V. Stojanovic. 2012. DSENT - A Tool Connecting Emerging Photonics with Electronics for Opto-Electronic Networks-on-Chip Modeling. In Proceedings of the International Symposium on Networks-on-Chip (NOCS). 201\u2013210."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2010.54"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CICC.2018.8357077"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2018.00074"},{"key":"e_1_3_2_1_36_1","volume-title":"Proceedings of International Symposium on Networks-on-Chip (NoCs). 64\u201373","author":"Wang Lei","year":"2009","unstructured":"Lei Wang, Yuho Jin, Hyungjun Kim, and Eun\u00a0Jung Kim. 2009. Recursive Partitioning Multicast: A Bandwidth-Efficient Routing for Networks-on-Chip. In Proceedings of International Symposium on Networks-on-Chip (NoCs). 64\u201373."},{"key":"e_1_3_2_1_37_1","volume-title":"Proceedings of the International Parallel and Distributed Processing Symposium (IPDPS). 990\u2013999","author":"Wang L.","unstructured":"L. Wang, X. Zhao, D. Kaeli, Z. Wang, and L. Eeckhout. 2018. Intra-Cluster Coalescing to Reduce GPU NoC Pressure. In Proceedings of the International Parallel and Distributed Processing Symposium (IPDPS). 990\u2013999."},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the International Symposium on Computer Architecture (ISCA). 726\u2013738","author":"Yin J.","unstructured":"J. Yin, Z. Lin, O. Kayiran, M. Poremba, M. Shoaib Bin Altaf, N. Enright Jerger, and G.\u00a0H. Loh. 2018. Modular Routing Design for Chiplet-Based Systems. In Proceedings of the International Symposium on Computer Architecture (ISCA). 726\u2013738."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2018.00035"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2023.3237927"},{"key":"e_1_3_2_1_41_1","volume-title":"Zomaya and Young\u00a0Choon Lee","author":"Y.","year":"2012","unstructured":"Albert\u00a0Y. Zomaya and Young\u00a0Choon Lee. 2012. Energy-Efficient Distributed Computing Systems. Wiley-IEEE Computer Society Pr."}],"event":{"name":"ICPP '24: the 53rd International Conference on Parallel Processing","location":"Gotland Sweden","acronym":"ICPP '24"},"container-title":["Proceedings of the 53rd International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673038.3673075","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3673038.3673075","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,23]],"date-time":"2025-09-23T17:33:07Z","timestamp":1758648787000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673038.3673075"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,12]]},"references-count":41,"alternative-id":["10.1145\/3673038.3673075","10.1145\/3673038"],"URL":"https:\/\/doi.org\/10.1145\/3673038.3673075","relation":{},"subject":[],"published":{"date-parts":[[2024,8,12]]},"assertion":[{"value":"2024-08-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}