{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:33:29Z","timestamp":1750221209424,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":31,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,9,23]],"date-time":"2018-09-23T00:00:00Z","timestamp":1537660800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,9,23]]},"DOI":"10.1145\/3236367.3236371","type":"proceedings-article","created":{"date-parts":[[2018,9,19]],"date-time":"2018-09-19T12:16:51Z","timestamp":1537359411000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Multi-Threading and Lock-Free MPI RMA Based Graph Processing on KNL and POWER Architectures"],"prefix":"10.1145","author":[{"given":"Mingzhe","family":"Li","sequence":"first","affiliation":[{"name":"The Ohio State University, Columbus, Ohio"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoyi","family":"Lu","sequence":"additional","affiliation":[{"name":"The Ohio State University, Columbus, Ohio"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hari","family":"Subramoni","sequence":"additional","affiliation":[{"name":"The Ohio State University, Columbus, Ohio"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dhabaleswar K.","family":"Panda","sequence":"additional","affiliation":[{"name":"The Ohio State University, Columbus, Ohio"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2018,9,23]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2017. OpenPOWER. https:\/\/openpowerfoundation.org\/. (2017).  2017. OpenPOWER. https:\/\/openpowerfoundation.org\/. (2017)."},{"key":"e_1_3_2_1_2_1","unstructured":"2017. POWER8 -- The First OpenPOWER Processor https:\/\/openpowerfoundation.org\/blogs\/power8-the-first-openpower-processor\/. (2017).  2017. POWER8 -- The First OpenPOWER Processor https:\/\/openpowerfoundation.org\/blogs\/power8-the-first-openpower-processor\/. (2017)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2010.46"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2006.34"},{"volume-title":"Optimizing Bandwidth Limited Problems Using One-sided Communication and Overlap. In IEEE International Parallel and Distributed Processing Symposium, 2006. IPDPS 2006. 1--10","author":"Bell C.","key":"e_1_3_2_1_5_1","unstructured":"C. Bell , D. Bonachea , R. Nishala , and K. Yelick . 2006 . Optimizing Bandwidth Limited Problems Using One-sided Communication and Overlap. In IEEE International Parallel and Distributed Processing Symposium, 2006. IPDPS 2006. 1--10 . C. Bell, D. Bonachea, R. Nishala, and K. Yelick. 2006. Optimizing Bandwidth Limited Problems Using One-sided Communication and Overlap. In IEEE International Parallel and Distributed Processing Symposium, 2006. IPDPS 2006. 1--10."},{"volume-title":"SlimSell: A Vectorizable Graph Representation for Breadth-First Search. In 2017 IEEE International Parallel and Distributed Processing Symposium (IPDPS). 32--41","author":"Besta M.","key":"e_1_3_2_1_6_1","unstructured":"M. Besta , F. Marending , E. Solomonik , and T. Hoefler . 2017 . SlimSell: A Vectorizable Graph Representation for Breadth-First Search. In 2017 IEEE International Parallel and Distributed Processing Symposium (IPDPS). 32--41 . M. Besta, F. Marending, E. Solomonik, and T. Hoefler. 2017. SlimSell: A Vectorizable Graph Representation for Breadth-First Search. In 2017 IEEE International Parallel and Distributed Processing Symposium (IPDPS). 32--41."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/2063384.2063471"},{"key":"e_1_3_2_1_8_1","unstructured":"TACC Stampede KNL Cluster. 2017. https:\/\/portal.tacc.utexas.edu\/user-guides\/stampede. (2017).  TACC Stampede KNL Cluster. 2017. https:\/\/portal.tacc.utexas.edu\/user-guides\/stampede. (2017)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2010.26"},{"volume-title":"International Conference on High Performance Computing, Student Research Symposium. IEEE.","author":"Edmonds N.","key":"e_1_3_2_1_10_1","unstructured":"N. Edmonds , J. Willock , T. Hoefler , and A. Lumsdaine . 2010. Design of a Large-Scale Hybrid-Parallel Graph Library . In International Conference on High Performance Computing, Student Research Symposium. IEEE. N. Edmonds, J. Willock, T. Hoefler, and A. Lumsdaine. 2010. Design of a Large-Scale Hybrid-Parallel Graph Library. In International Conference on High Performance Computing, Student Research Symposium. IEEE."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/2503210.2503286"},{"key":"e_1_3_2_1_12_1","unstructured":"The Graph500. 2017. http:\/\/www.graph500.org. (2017).  The Graph500. 2017. http:\/\/www.graph500.org. (2017)."},{"key":"e_1_3_2_1_13_1","volume":"200","author":"Harish Pawan","unstructured":"Pawan Harish and P. J. Narayanan. 200 7. Accelerating Large Graph Algorithms on the GPU Using CUDA. In Proceedings of the 14th International Conference on High Performance Computing (HiPC'07). 197--208. Pawan Harish and P. J. Narayanan. 2007. Accelerating Large Graph Algorithms on the GPU Using CUDA. In Proceedings of the 14th International Conference on High Performance Computing (HiPC'07). 197--208.","journal-title":"J. Narayanan."},{"key":"e_1_3_2_1_14_1","volume-title":"Moreira","author":"Karkhanis Tejas S.","year":"2011","unstructured":"Tejas S. Karkhanis and Jos\u00e9 E . Moreira . 2011 . \"IBM Power Architecture\". Springer US , Boston, MA, 900--907. Tejas S. Karkhanis and Jos\u00e9 E. Moreira. 2011. \"IBM Power Architecture\". Springer US, Boston, MA, 900--907."},{"key":"e_1_3_2_1_15_1","unstructured":"Network Based Computing Laboratory. 2017. OSU Micro-benchmarks. http:\/\/mvapich.cse.ohio-state.edu\/benchmarks. (2017).  Network Based Computing Laboratory. 2017. OSU Micro-benchmarks. http:\/\/mvapich.cse.ohio-state.edu\/benchmarks. (2017)."},{"volume-title":"Using Kronecker Multiplication. In Conference on Principles and Practice of Knowledge Discovery in Databases.","author":"Leskovec J.","key":"e_1_3_2_1_16_1","unstructured":"J. Leskovec , D. Chakrabarti , J. Kleinberg , and C. Faloutsos . 2005. Realistic, Mathematically Tractable Graph Generation and Evolution , Using Kronecker Multiplication. In Conference on Principles and Practice of Knowledge Discovery in Databases. J. Leskovec, D. Chakrabarti, J. Kleinberg, and C. Faloutsos. 2005. Realistic, Mathematically Tractable Graph Generation and Evolution, Using Kronecker Multiplication. In Conference on Principles and Practice of Knowledge Discovery in Databases."},{"key":"e_1_3_2_1_17_1","volume-title":"(DK) Panda","author":"Li Mingzhe","year":"2015","unstructured":"Mingzhe Li , Khaled Hamidouche , Xiaoyi Lu , Jian Lin , and Dhabaleswar K . (DK) Panda . 2015 . High-Performance and Scalable Design of MPI-3 RMA on Xeon Phi Clusters . Mingzhe Li, Khaled Hamidouche, Xiaoyi Lu, Jian Lin, and Dhabaleswar K. (DK) Panda. 2015. High-Performance and Scalable Design of MPI-3 RMA on Xeon Phi Clusters."},{"volume-title":"2014 IEEE International Conference on Cluster Computing (CLUSTER). 230--238","author":"Li M.","key":"e_1_3_2_1_18_1","unstructured":"M. Li , X. Lu , S. Potluri , K. Hamidouche , J. Jose , K. Tomko , and D. K. Panda . 2014. Scalable Graph500 Design with MPI-3 RMA . In 2014 IEEE International Conference on Cluster Computing (CLUSTER). 230--238 . M. Li, X. Lu, S. Potluri, K. Hamidouche, J. Jose, K. Tomko, and D. K. Panda. 2014. Scalable Graph500 Design with MPI-3 RMA. In 2014 IEEE International Conference on Cluster Computing (CLUSTER). 230--238."},{"key":"e_1_3_2_1_19_1","unstructured":"X. Liu L. Chen J. S. Firoz J. Qiu and L. Jiang. 2017. Performance Characterization of Multithreaded Graph Processing Applications on Intel Many-Integrated-Core Architecture. ArXiv e-prints (Aug. 2017).  X. Liu L. Chen J. S. Firoz J. Qiu and L. Jiang. 2017. Performance Characterization of Multithreaded Graph Processing Applications on Intel Many-Integrated-Core Architecture. ArXiv e-prints (Aug. 2017)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/1837274.1837289"},{"volume-title":"UPC on MIC: Early Experiences with Native and Symmetric Modes. In 7th International Conference on PGAS Programming Models (PGAS).","author":"Luo M.","key":"e_1_3_2_1_21_1","unstructured":"M. Luo , M. Li , A. Venkatesh , X. Lu , and D. K. Panda . 2013 . UPC on MIC: Early Experiences with Native and Symmetric Modes. In 7th International Conference on PGAS Programming Models (PGAS). M. Luo, M. Li, A. Venkatesh, X. Lu, and D. K. Panda. 2013. UPC on MIC: Early Experiences with Native and Symmetric Modes. In 7th International Conference on PGAS Programming Models (PGAS)."},{"key":"e_1_3_2_1_22_1","unstructured":"Memkind. 2017. https:\/\/github.com\/memkind\/memkind. (2017).  Memkind. 2017. https:\/\/github.com\/memkind\/memkind. (2017)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2009.5161108"},{"key":"e_1_3_2_1_24_1","unstructured":"nvGRAPH. 2017. https:\/\/devblogs.nvidia.com\/parallelforall\/cuda-8-features-revealed\/. (2017).  nvGRAPH. 2017. https:\/\/devblogs.nvidia.com\/parallelforall\/cuda-8-features-revealed\/. (2017)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/1810085.1810092"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2011.6114175"},{"key":"e_1_3_2_1_27_1","unstructured":"Cori System. 2017. http:\/\/www.nersc.gov\/users\/computational-systems\/cori\/. (2017).  Cori System. 2017. http:\/\/www.nersc.gov\/users\/computational-systems\/cori\/. (2017)."},{"key":"e_1_3_2_1_28_1","unstructured":"Coral System. 2017. https:\/\/asc.llnl.gov\/coral-info. (2017).  Coral System. 2017. https:\/\/asc.llnl.gov\/coral-info. (2017)."},{"key":"e_1_3_2_1_29_1","unstructured":"Oakforest-PACS System. 2017. http:\/\/www.cc.u-tokyo.ac.jp\/system\/ofp\/.(2017).  Oakforest-PACS System. 2017. http:\/\/www.cc.u-tokyo.ac.jp\/system\/ofp\/.(2017)."},{"volume-title":"Proceedings of 21st International Conference on Parallel and Distributed Computing Systems (PDCS'09)","author":"Xia Y.","key":"e_1_3_2_1_30_1","unstructured":"Y. Xia and V.K. Prasanna . 2009. Topologically Adaptive Parallel Breadth-First Search on Multicore Processors . In Proceedings of 21st International Conference on Parallel and Distributed Computing Systems (PDCS'09) . Y. Xia and V.K. Prasanna. 2009. Topologically Adaptive Parallel Breadth-First Search on Multicore Processors. In Proceedings of 21st International Conference on Parallel and Distributed Computing Systems (PDCS'09)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2005.4"}],"event":{"name":"EuroMPI'18: 25th European MPI Users' Group Meeting","acronym":"EuroMPI'18","location":"Barcelona Spain"},"container-title":["Proceedings of the 25th European MPI Users' Group Meeting"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3236367.3236371","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3236367.3236371","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T01:39:39Z","timestamp":1750210779000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3236367.3236371"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,9,23]]},"references-count":31,"alternative-id":["10.1145\/3236367.3236371","10.1145\/3236367"],"URL":"https:\/\/doi.org\/10.1145\/3236367.3236371","relation":{},"subject":[],"published":{"date-parts":[[2018,9,23]]},"assertion":[{"value":"2018-09-23","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}