{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T04:20:37Z","timestamp":1750306837840,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","license":[{"start":{"date-parts":[[2014,6,10]],"date-time":"2014-06-10T00:00:00Z","timestamp":1402358400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2014,6,10]]},"DOI":"10.1145\/2597652.2597654","type":"proceedings-article","created":{"date-parts":[[2014,6,10]],"date-time":"2014-06-10T12:50:25Z","timestamp":1402404625000},"page":"343-352","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Collective memory transfers for multi-core chips"],"prefix":"10.1145","author":[{"given":"George","family":"Michelogiannakis","sequence":"first","affiliation":[{"name":"Lawrence Berkeley National Laboratory, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alexander","family":"Williams","sequence":"additional","affiliation":[{"name":"Lawrence Berkeley National Laboratory, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Samuel","family":"Williams","sequence":"additional","affiliation":[{"name":"Lawrence Berkeley National Laboratory, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"John","family":"Shalf","sequence":"additional","affiliation":[{"name":"Lawrence Berkeley National Laboratory, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2014,6,10]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"VECPAR","author":"A.","year":"2012","unstructured":"A. Abdelfattah phet al., \"Optimizing memory-bound numerical kernels on GPU hardware accelerators,\" ser . VECPAR , Kobe, Japan , 2012 . A. Abdelfattah phet al., \"Optimizing memory-bound numerical kernels on GPU hardware accelerators,\" ser. VECPAR, Kobe, Japan, 2012."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/1555754.1555810"},{"key":"e_1_3_2_1_3_1","volume-title":"dissertation","author":"Bienia C.","year":"2011","unstructured":"C. Bienia , \"Benchmarking modern multiprocessors,\" Ph. D. dissertation , Princeton University , January 2011 . C. Bienia, \"Benchmarking modern multiprocessors,\" Ph.D. dissertation, Princeton University, January 2011."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/1375581.1375595"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/1941487.1941507"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2011.73"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2008.917729"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2004.836009"},{"key":"e_1_3_2_1_9_1","volume-title":"Principles and Practices of Interconnection Networks","author":"Dally W. J.","year":"2003","unstructured":"W. J. Dally and B. Towles , Principles and Practices of Interconnection Networks . Morgan Kaufmann Publishers Inc ., 2003 . W. J. Dally and B. Towles, Principles and Practices of Interconnection Networks. Morgan Kaufmann Publishers Inc., 2003."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"K. Datta phet al. \"Stencil computation optimization and auto-tuning on state-of-the-art multicore architectures \" ser. SC 2008.   K. Datta phet al. \"Stencil computation optimization and auto-tuning on state-of-the-art multicore architectures \" ser. SC 2008.","DOI":"10.1109\/SC.2008.5222004"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/2155620.2155663"},{"key":"e_1_3_2_1_12_1","volume-title":"ICCSIT","volume":"5","author":"W.","year":"2010","unstructured":"W. Gao phet al., \"An improved sobel edge detection,\" ser . ICCSIT , vol. 5 , 2010 . W. Gao phet al., \"An improved sobel edge detection,\" ser. ICCSIT, vol. 5, 2010."},{"key":"e_1_3_2_1_13_1","volume-title":"Virtual Vector Architecture,\" ser. Berlin","author":"J.","year":"2009","unstructured":"J. Gebis phet al., \"Improving memory subsystem performance using ViVA : Virtual Vector Architecture,\" ser. Berlin , Heidelberg : Springer-Verlag , 2009 . J. Gebis phet al., \"Improving memory subsystem performance using ViVA: Virtual Vector Architecture,\" ser. Berlin, Heidelberg: Springer-Verlag, 2009."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/1810085.1810111"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.v21:1"},{"key":"e_1_3_2_1_16_1","volume-title":"CC\/ETAPS","author":"T.","year":"2011","unstructured":"T. Henretty phet al., \"Data layout transformation for stencil computations on short-vector SIMD architectures,\" ser . CC\/ETAPS , 2011 . T. Henretty phet al., \"Data layout transformation for stencil computations on short-vector SIMD architectures,\" ser. CC\/ETAPS, 2011."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/2304576.2304619"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/2304576.2304614"},{"key":"e_1_3_2_1_20_1","volume-title":"IPDPS","author":"S.","year":"2010","unstructured":"S. Kamil phet al., \"An auto-tuning framework for parallel multicore stencil computations,\" ser . IPDPS , 2010 . S. Kamil phet al., \"An auto-tuning framework for parallel multicore stencil computations,\" ser. IPDPS, 2010."},{"key":"e_1_3_2_1_21_1","volume-title":"DATE '09","author":"Kandemir M.","year":"2009","unstructured":"M. Kandemir , Y. Zhang , and O. Ozturk , \" Adaptive prefetching for shared cache based chip multiprocessors,\" ser . DATE '09 , 2009 . M. Kandemir, Y. Zhang, and O. Ozturk, \"Adaptive prefetching for shared cache based chip multiprocessors,\" ser. DATE '09, 2009."},{"key":"e_1_3_2_1_22_1","volume-title":"NSS\/MIC '12","author":"Kavianipour H.","year":"2012","unstructured":"H. Kavianipour and C. Bohm , \" High performance FPGA-based scatter\/gather DMA interface for PCIe,\" ser . NSS\/MIC '12 , 2012 . H. Kavianipour and C. Bohm, \"High performance FPGA-based scatter\/gather DMA interface for PCIe,\" ser. NSS\/MIC '12, 2012."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.5555\/874077.876492"},{"key":"e_1_3_2_1_24_1","volume-title":"IPDPS","author":"G.","year":"2008","unstructured":"G. Khanna phet al., \"A dynamic scheduling approach for coordinated wide-area data transfers using GridFTP,\" ser . IPDPS , 2008 . G. Khanna phet al., \"A dynamic scheduling approach for coordinated wide-area data transfers using GridFTP,\" ser. IPDPS, 2008."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2003.1261385"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/2063384.2063482"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/1393921.1393967"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/1654059.1654072"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2008.7"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2009.5161011"},{"key":"e_1_3_2_1_31_1","volume-title":"dissertation","author":"Rixner S.","year":"2001","unstructured":"S. Rixner , \" A bandwidth-efficient architecture for a streaming media processor,\" Ph. D. dissertation , Massachusetts Institute of Technology , 2001 . S. Rixner, \"A bandwidth-efficient architecture for a streaming media processor,\" Ph.D. dissertation, Massachusetts Institute of Technology, 2001."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/339647.339668"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/1555754.1555801"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/L-CA.2011.4"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/359327.359336"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/MC.2009.407"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/224170.224371"},{"key":"e_1_3_2_1_38_1","volume-title":"VECPAR","author":"Shalf J.","year":"2010","unstructured":"J. Shalf , S. S. Dosanjh , and J. Morrison , \" Exascale computing technology challenges,\" ser . VECPAR , 2010 . J. Shalf, S. S. Dosanjh, and J. Morrison, \"Exascale computing technology challenges,\" ser. VECPAR, 2010."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/1815961.1815972"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2013.6522356"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/1736020.1736045"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/1815961.1815983"},{"key":"e_1_3_2_1_43_1","volume-title":"SAMOS","author":"A.","year":"2011","unstructured":"A. Vega phet al., \"Breaking the bandwidth wall in chip multiprocessors,\" ser . SAMOS , 2011 . A. Vega phet al., \"Breaking the bandwidth wall in chip multiprocessors,\" ser. SAMOS, 2011."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/1669112.1669119"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/12.966490"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/2259016.2259044"}],"event":{"name":"ICS'14: 2014 International Conference on Supercomputing","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"],"location":"Munich Germany","acronym":"ICS'14"},"container-title":["Proceedings of the 28th ACM international conference on Supercomputing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2597652.2597654","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2597652.2597654","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T08:09:58Z","timestamp":1750234198000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2597652.2597654"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,6,10]]},"references-count":45,"alternative-id":["10.1145\/2597652.2597654","10.1145\/2597652"],"URL":"https:\/\/doi.org\/10.1145\/2597652.2597654","relation":{},"subject":[],"published":{"date-parts":[[2014,6,10]]},"assertion":[{"value":"2014-06-10","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}