{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,23]],"date-time":"2025-09-23T18:36:08Z","timestamp":1758652568595,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":41,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,12]],"date-time":"2024-08-12T00:00:00Z","timestamp":1723420800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"National Key R&D Program of China","award":["2020YFA0607903"],"award-info":[{"award-number":["2020YFA0607903"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,12]]},"DOI":"10.1145\/3673038.3673131","type":"proceedings-article","created":{"date-parts":[[2024,8,8]],"date-time":"2024-08-08T18:29:01Z","timestamp":1723141741000},"page":"262-272","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["BoostN: Optimizing Imbalanced Neighborhood Communication on Homogeneous Many-Core System"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-1786-9821","authenticated-orcid":false,"given":"Haopeng","family":"Huang","sequence":"first","affiliation":[{"name":"Tsinghua University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2358-3395","authenticated-orcid":false,"given":"Yuyang","family":"Jin","sequence":"additional","affiliation":[{"name":"Tsinghua University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9740-6581","authenticated-orcid":false,"given":"Wei","family":"Xue","sequence":"additional","affiliation":[{"name":"Tsinghua University, China; Qinghai University, China and Qinghai University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,8,12]]},"reference":[{"volume-title":"AMG2013","year":"2013","key":"e_1_3_2_1_1_1","unstructured":"2023. AMG2013. https:\/\/asc.llnl.gov\/codes\/proxy-apps\/amg2013 [Accessed 23-06-2024]."},{"key":"e_1_3_2_1_2_1","unstructured":"2023. HYPRE. https:\/\/computing.llnl.gov\/projects\/hypre-scalable-linear-solvers-multigrid-methods. [Accessed 23-06-2024]."},{"key":"e_1_3_2_1_3_1","unstructured":"2024. AMD EPYC 9754. https:\/\/www.amd.com\/en\/products\/cpu\/amd-epyc-9754. [Accessed 23-06-2024]."},{"key":"e_1_3_2_1_4_1","unstructured":"2024. HPC-X. https:\/\/developer.nvidia.com\/networking\/. [Accessed 23-06-2024]."},{"key":"e_1_3_2_1_5_1","unstructured":"2024. MPI standard. https:\/\/www.mpi-forum.org\/ [Accessed 23-06-2024]."},{"key":"e_1_3_2_1_6_1","unstructured":"2024. MPICH | High-Performance Portable MPI \u2014 mpich.org. https:\/\/www.mpich.org\/. [Accessed 23-06-2024]."},{"key":"e_1_3_2_1_7_1","unstructured":"2024. Open MPI: Open Source High Performance Computing \u2014 open-mpi.org. https:\/\/www.open-mpi.org\/. [Accessed 23-06-2024]."},{"key":"e_1_3_2_1_8_1","unstructured":"2024. SuiteSparse Matrix Collection. https:\/\/sparse.tamu.edu\/. [Accessed 23-06-2024]."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/215399.215427"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.camwa.2020.06.009"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Satish Balay Shrirang Abhyankar Mark Adams Jed Brown Peter Brune Kris Buschelman Lisandro Dalcin Alp Dener Victor Eijkhout William Gropp 2019. PETSc users manual. (2019).","DOI":"10.2172\/1577437"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342020925535"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00031"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/1183401.1183451"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3624062.3624111"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/155332.155333"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1029\/2019MS001916"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1137\/130931539"},{"key":"e_1_3_2_1_19_1","volume-title":"DASH: Data structures and algorithms with support for hierarchical locality. In Euro-Par 2014: Parallel Processing Workshops: Euro-Par 2014 International Workshops","author":"F\u00fcrlinger Karl","year":"2014","unstructured":"Karl F\u00fcrlinger, Colin Glass, Jose Gracia, Andreas Kn\u00fcpfer, Jie Tao, Denis H\u00fcnich, Kamran Idrees, Matthias Maiterth, Yousri Mhedheb, and Huan Zhou. 2014. DASH: Data structures and algorithms with support for hierarchical locality. In Euro-Par 2014: Parallel Processing Workshops: Euro-Par 2014 International Workshops, Porto, Portugal, August 25-26, 2014, Revised Selected Papers, Part II 20. Springer, 542\u2013552."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/1995896.1995924"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2012.09.016"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/BFb0056575"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Michael\u00a0A Heroux Lois\u00a0Curfman McInnes Rajeev Thakur Jeffrey\u00a0S Vetter Xiaoye\u00a0Sherry Li James Aherns Todd Munson and Kathryn Mohror. 2020. ECP software technology capability assessment report. Technical Report. Oak Ridge National Lab.(ORNL) Oak Ridge TN (United States).","DOI":"10.2172\/1760096"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/1995896.1995909"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTR.2008.4663761"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/165854.165874"},{"key":"e_1_3_2_1_27_1","volume-title":"METIS: Unstructured graph partitioning and sparse matrix ordering system. Technical report","author":"Karypis George","year":"1997","unstructured":"George Karypis. 1997. METIS: Unstructured graph partitioning and sparse matrix ordering system. Technical report (1997)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.5555\/305219.305248"},{"key":"e_1_3_2_1_29_1","volume-title":"Stream benchmark. Link: www. cs. virginia. edu\/stream\/ref. html# what 22, 7","author":"McCalpin D","year":"1995","unstructured":"John\u00a0D McCalpin. 1995. Stream benchmark. Link: www. cs. virginia. edu\/stream\/ref. html# what 22, 7 (1995)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2008.09.001"},{"volume-title":"ACM Sigplan Fortran Forum, Vol.\u00a017. ACM New York","author":"Numrich W","key":"e_1_3_2_1_31_1","unstructured":"Robert\u00a0W Numrich and John Reid. 1998. Co-Array Fortran for parallel programming. In ACM Sigplan Fortran Forum, Vol.\u00a017. ACM New York, NY, USA, 1\u201331."},{"key":"e_1_3_2_1_32_1","volume-title":"Xpmem: Cross-process memory mapping.","author":"Pedretti K","year":"2020","unstructured":"K Pedretti and B Barrett. 2020. Xpmem: Cross-process memory mapping."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607074"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627535.3638503"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126963"},{"key":"e_1_3_2_1_36_1","volume-title":"Proceedings of the 16th International Symposium on Experimental Algorithms (SEA\u201917)","author":"Schulz Christian","year":"2017","unstructured":"Christian Schulz and Jesper\u00a0Larsson Tr\u00e4ff. 2017. Better Process Mapping and Sparse Quadratic Assignment. In Proceedings of the 16th International Symposium on Experimental Algorithms (SEA\u201917)(LIPIcs, Vol.\u00a075). Dagstuhl, 4:1 \u2013 4:15. Technical Report, arXiv:1702.04164."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/1542275.1542320"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cpc.2021.108171"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1137\/S1064827598337373"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/1654059.1654087"},{"volume-title":"a PGAS extension for C++. In 2014 IEEE 28th international parallel and distributed processing symposium","author":"Zheng Yili","key":"e_1_3_2_1_41_1","unstructured":"Yili Zheng, Amir Kamil, Michael\u00a0B Driscoll, Hongzhang Shan, and Katherine Yelick. 2014. UPC++: a PGAS extension for C++. In 2014 IEEE 28th international parallel and distributed processing symposium. IEEE, 1105\u20131114."}],"event":{"name":"ICPP '24: the 53rd International Conference on Parallel Processing","acronym":"ICPP '24","location":"Gotland Sweden"},"container-title":["Proceedings of the 53rd International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673038.3673131","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3673038.3673131","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,23]],"date-time":"2025-09-23T17:31:09Z","timestamp":1758648669000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673038.3673131"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,12]]},"references-count":41,"alternative-id":["10.1145\/3673038.3673131","10.1145\/3673038"],"URL":"https:\/\/doi.org\/10.1145\/3673038.3673131","relation":{},"subject":[],"published":{"date-parts":[[2024,8,12]]},"assertion":[{"value":"2024-08-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}