{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:28:29Z","timestamp":1750220909034,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,6,23]],"date-time":"2020-06-23T00:00:00Z","timestamp":1592870400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,6,23]]},"DOI":"10.1145\/3369583.3392670","type":"proceedings-article","created":{"date-parts":[[2020,6,22]],"date-time":"2020-06-22T03:27:27Z","timestamp":1592796447000},"page":"137-148","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["PAC: Paged Adaptive Coalescer for 3D-Stacked Memory"],"prefix":"10.1145","author":[{"given":"Xi","family":"Wang","sequence":"first","affiliation":[{"name":"Texas Tech University, Lubbock, TX, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"John D.","family":"Leidel","sequence":"additional","affiliation":[{"name":"Tactical Computing Labs, Muenster, TX, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Brody","family":"Williams","sequence":"additional","affiliation":[{"name":"Texas Tech University, Lubbock, TX, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yong","family":"Chen","sequence":"additional","affiliation":[{"name":"Texas Tech University, Lubbock, TX, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2020,6,23]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"A throughput-optimized optical network for data-intensive computing","author":"Laurent Schares","year":"2014","unstructured":"Laurent Schares et al. A throughput-optimized optical network for data-intensive computing. IEEE Micro, 2014."},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2017.54"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750385"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2015.22"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2016.7446059"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCS.2018.00061"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/2445572.2445574"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2010.44"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2010.107"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/2830772.2830830"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2016.7446089"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2012.32"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2014.6835964"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/2540708.2540741"},{"volume-title":"SC 2011.","author":"Che Shuai","key":"e_1_3_2_2_15_1","unstructured":"Shuai Che, Jeremy W Sheaffer, and Kevin Skadron. Dymaxion: Optimizing memory access patterns for heterogeneous systems. In SC 2011."},{"key":"e_1_3_2_2_16_1","volume-title":"December","author":"Cube Hybrid Memory","year":"2015","unstructured":"Hybrid Memory Cube Specification 2.1. Technical report, December 2015."},{"key":"e_1_3_2_2_17_1","volume-title":"Aga and Satish Narayanasamy. InvisiMem: Smart Memory Defenses for Memory Bus Side Channel. In ISCA","author":"Shaizeen","year":"2017","unstructured":"Shaizeen Aga and Satish Narayanasamy. InvisiMem: Smart Memory Defenses for Memory Bus Side Channel. In ISCA 2017."},{"key":"e_1_3_2_2_18_1","first-page":"544","volume-title":"Xuehai Qian. GraphP: Reducing Communication for PIM-Based Graph Processing with Efficient Data Partition. In HPCA","author":"Zhang Mingxing","year":"2018","unstructured":"Mingxing Zhang, Youwei Zhuo, Chao Wang, Mingyu Gao, Yongwei Wu, Kang Chen, Christos Kozyrakis, and Xuehai Qian. GraphP: Reducing Communication for PIM-Based Graph Processing with Efficient Data Partition. In HPCA 2018, pages 544--557. IEEE."},{"key":"e_1_3_2_2_19_1","volume-title":"PACT","author":"Kim Gwangsun","year":"2013","unstructured":"Gwangsun Kim, John Kim, Jung Ho Ahn, and Jaeha Kim. Memory-centric system interconnect design with hybrid memory cubes. In PACT 2013."},{"key":"e_1_3_2_2_21_1","volume-title":"Kartikay Garg, Tushar Krishna, and Hyesoon Kim. Performance Implications of NoCs on 3D-Stacked Memories: Insights from the Hybrid Memory Cube. arXiv preprint arXiv:1707.05399","author":"Hadidi Ramyad","year":"2017","unstructured":"Ramyad Hadidi, Bahar Asgari, Jeffrey Young, Burhan Ahmad Mudassar, Kartikay Garg, Tushar Krishna, and Hyesoon Kim. Performance Implications of NoCs on 3D-Stacked Memories: Insights from the Hybrid Memory Cube. arXiv preprint arXiv:1707.05399, 2017."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/2833179.2833184"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/HOTCHIPS.2011.7477494"},{"key":"e_1_3_2_2_24_1","volume-title":"EECS Department","author":"Waterman Andrew","year":"2015","unstructured":"Andrew Waterman, Yunsup Lee, Rimas Avizienis, David A. Patterson, and Krste Asanovi?. The RISC-V Instruction Set Manual Volume II: Privileged Architecture Version 1.7. Technical report, EECS Department, University of California, Berkeley, 2015."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/2000064.2000100"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/2464996.2465019"},{"key":"e_1_3_2_2_27_1","volume-title":"ISCA","author":"Kroft David","year":"1981","unstructured":"David Kroft. Lockup-free instruction fetch\/prefetch cache organization. In ISCA 1981."},{"key":"e_1_3_2_2_28_1","volume-title":"July","author":"Toolkit Documentation CUDA","year":"2018","unstructured":"CUDA Toolkit Documentation. Technical report, July 2018."},{"key":"e_1_3_2_2_29_1","volume-title":"HPCA","author":"Power Jason","year":"2014","unstructured":"Jason Power, Mark D Hill, and David A Wood. Supporting x86--64 address translation for 100s of gpu lanes. In HPCA 2014."},{"key":"e_1_3_2_2_30_1","volume-title":"Hot Chips","volume":"30","author":"Yoshida Toshio","year":"2018","unstructured":"Toshio Yoshida. Fujitsu high performance cpu for the post-k computer. In Hot Chips, volume 30, 2018."},{"key":"e_1_3_2_2_31_1","volume-title":"ISCA","author":"Rixner S","year":"2000","unstructured":"S Rixner, WJ Dally, UJ Kapasi, P Mattson, and JD Owens. Memory access scheduling. In ISCA 2000."},{"key":"e_1_3_2_2_32_1","first-page":"62","volume-title":"Yong Chen. Memory Coalescing for Hybrid Memory Cube. In ICPP","author":"Wang Xi","year":"2018","unstructured":"Xi Wang, John D Leidel, and Yong Chen. Memory Coalescing for Hybrid Memory Cube. In ICPP 2018, page 62. ACM."},{"key":"e_1_3_2_2_33_1","first-page":"1","volume-title":"Proceedings of the 48th International Conference on Parallel Processing","author":"Wang Xi","year":"2019","unstructured":"Xi Wang, Antonino Tumeo, John D Leidel, Jie Li, and Yong Chen. Mac: Memory access coalescer for 3d-stacked memory. In Proceedings of the 48th International Conference on Parallel Processing, pages 1--10, 2019."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2017.58"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/VLSIT.2012.6242474"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2017.8167757"},{"key":"e_1_3_2_2_37_1","volume":"201","author":"Ravi Nair","unstructured":"Ravi Nair et al. Active memory cube: A processing-in-memory architecture for exascale systems. IBM Journal of Research and Development, 2015.","journal-title":"IBM Journal of Research and Development"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2006.44"},{"key":"e_1_3_2_2_39_1","volume-title":"A Survey of Memory Bandwidth and Machine Balance in Current High Performance Computers","author":"McCalpin John D.","year":"1995","unstructured":"John D. McCalpin. A Survey of Memory Bandwidth and Machine Balance in Current High Performance Computers, 1995."},{"key":"e_1_3_2_2_40_1","volume-title":"Design and Implementation of the HPCS Graph Analysis Benchmark on Symmetric Multiprocessors. HiPC","author":"Bader David","year":"2005","unstructured":"David Bader and Kamesh Madduri. Design and Implementation of the HPCS Graph Analysis Benchmark on Symmetric Multiprocessors. HiPC 2005."},{"key":"e_1_3_2_2_41_1","volume-title":"Sandia National Laboratories","author":"New Toward","year":"2013","unstructured":"Toward a New Metric for Ranking High Performance Computing Systems. Technical report, Sandia National Laboratories, 2013."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2009.64"},{"key":"e_1_3_2_2_43_1","volume-title":"NAS Parallel Benchmark Results. In SC 1992","author":"Bailey D. H.","year":"1992","unstructured":"D. H. Bailey, L. Dagum, E. Barszcz, and H. D. Simon. NAS Parallel Benchmark Results. In SC 1992, Supercomputing 1992."},{"key":"e_1_3_2_2_44_1","volume-title":"CoRR","author":"Beamer Scott","year":"2015","unstructured":"Scott Beamer, Krste Asanovic, and David A. Patterson. The GAP benchmark suite. CoRR, 2015."},{"key":"e_1_3_2_2_45_1","volume-title":"PLDI","author":"Chen Dong","year":"2019","unstructured":"Dong Chen, Fangzhou Liu, Chen Ding, and Sreepathi Pai. Locality analysis through static parallel sampling. In PLDI 2019."},{"key":"e_1_3_2_2_46_1","volume-title":"ISSCC","author":"Shin Dongjoo","year":"2017","unstructured":"Dongjoo Shin, Jinmook Lee, Jinsu Lee, and Hoi-Jun Yoo. 14.2 DNPU: An 8.1 TOPS\/W reconfigurable CNN-RNN processor for general-purpose deep neural networks. In ISSCC 2017."},{"key":"e_1_3_2_2_47_1","volume-title":"Technical report","author":"Standard High Bandwidth JEDEC","year":"2013","unstructured":"JEDEC Standard High Bandwidth Memory(HBM) DRAM Specification. Technical report, 2013."},{"key":"e_1_3_2_2_48_1","volume-title":"MICRO","author":"O'Connor Mike","year":"2017","unstructured":"Mike O'Connor, Niladrish Chatterjee, Donghyuk Lee, John Wilson, Aditya Agrawal, Stephen W Keckler, and William J Dally. Fine-grained DRAM: energy-efficient DRAM for extreme bandwidth systems. In MICRO 2017."},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2014.6835972"},{"key":"e_1_3_2_2_50_1","volume-title":"June","author":"Sunny","year":"2015","unstructured":"Sunny G. Using The AutoHBW Library with Jemalloc and Memkind. Technical report, June 2015."},{"key":"e_1_3_2_2_51_1","volume-title":"September","author":"Alberto","year":"2016","unstructured":"Alberto V. Improve Vectorization Performance with Intel? AVX-512. Technical report, September 2016."},{"key":"e_1_3_2_2_52_1","volume":"201","author":"Leidel John D","unstructured":"John D Leidel and Yong Chen. HMC-Sim: A simulation framework for hybrid memory cube devices. Parallel Processing Letters, 2014.","journal-title":"Parallel Processing Letters"},{"key":"e_1_3_2_2_53_1","volume-title":"Kdd","author":"Martin","year":"1996","unstructured":"Martin Ester et al. A density-based algorithm for discovering clusters in large spatial databases with noise. In Kdd, 1996."},{"volume-title":"Proceedings of the April 30--May 2, 1968, spring joint computer conference.","author":"Batcher Kenneth E","key":"e_1_3_2_2_54_1","unstructured":"Kenneth E Batcher. Sorting networks and their applications. In Proceedings of the April 30--May 2, 1968, spring joint computer conference."},{"key":"e_1_3_2_2_55_1","volume-title":"IPDPS","author":"Greb Alexander","year":"2006","unstructured":"Alexander Greb and Gabriel Zachmann. GPU-ABiSort: Optimal parallel sorting on stream architectures. In IPDPS 2006."},{"key":"e_1_3_2_2_56_1","volume-title":"IPDPS","author":"Ye Xiaochun","year":"2010","unstructured":"Xiaochun Ye, Dongrui Fan, Wei Lin, Nan Yuan, and Paolo Ienne. High performance comparison-based sorting algorithm on many-core GPUs. In IPDPS 2010."},{"key":"e_1_3_2_2_57_1","volume-title":"HPCA","author":"Hayes Timothy","year":"2015","unstructured":"Timothy Hayes, Oscar Palomar, Osman Unsal, Adrian Cristal, and Mateo Valero. VSR sort: A novel vectorised sorting algorithm & architecture extensions for future microprocessors. In HPCA 2015."}],"event":{"name":"HPDC '20: The 29th International Symposium on High-Performance Parallel and Distributed Computing","sponsor":["University of Arizona University of Arizona","SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"],"location":"Stockholm Sweden","acronym":"HPDC '20"},"container-title":["Proceedings of the 29th International Symposium on High-Performance Parallel and Distributed Computing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3369583.3392670","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3369583.3392670","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:44:58Z","timestamp":1750203898000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3369583.3392670"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,6,23]]},"references-count":56,"alternative-id":["10.1145\/3369583.3392670","10.1145\/3369583"],"URL":"https:\/\/doi.org\/10.1145\/3369583.3392670","relation":{},"subject":[],"published":{"date-parts":[[2020,6,23]]},"assertion":[{"value":"2020-06-23","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}