{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T23:50:49Z","timestamp":1783036249837,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":63,"publisher":"ACM","license":[{"start":{"date-parts":[[2015,6,13]],"date-time":"2015-06-13T00:00:00Z","timestamp":1434153600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000185","name":"Defense Advanced Research Projects Agency","doi-asserted-by":"publisher","award":["HR0011-13-2-0007"],"award-info":[{"award-number":["HR0011-13-2-0007"]}],"id":[{"id":"10.13039\/100000185","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2015,6,13]]},"DOI":"10.1145\/2749469.2750397","type":"proceedings-article","created":{"date-parts":[[2015,5,26]],"date-time":"2015-05-26T10:36:25Z","timestamp":1432636585000},"page":"131-143","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":108,"title":["Data reorganization in memory using 3D-stacked DRAM"],"prefix":"10.1145","author":[{"given":"Berkin","family":"Akin","sequence":"first","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Franz","family":"Franchetti","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"James C.","family":"Hoe","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2015,6,13]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"\"CACTI 6.5 HP labs \" http:\/\/www.hpl.hp.com\/research\/cacti\/.  \"CACTI 6.5 HP labs \" http:\/\/www.hpl.hp.com\/research\/cacti\/."},{"key":"e_1_3_2_1_2_1","unstructured":"\"DDR3-1600 dram datasheet MT41J256M4 Micron \" http:\/\/www.micron.com\/parts\/dram\/ddr3-sdram.  \"DDR3-1600 dram datasheet MT41J256M4 Micron \" http:\/\/www.micron.com\/parts\/dram\/ddr3-sdram."},{"key":"e_1_3_2_1_3_1","unstructured":"\"Intel math kernel library (MKL) \" http:\/\/software.intel.com\/en-us\/articles\/intel-mkl\/.  \"Intel math kernel library (MKL) \" http:\/\/software.intel.com\/en-us\/articles\/intel-mkl\/."},{"key":"e_1_3_2_1_4_1","unstructured":"\"McPAT 1.0 HP labs \" http:\/\/www.hpl.hp.com\/research\/mcpat\/.  \"McPAT 1.0 HP labs \" http:\/\/www.hpl.hp.com\/research\/mcpat\/."},{"key":"e_1_3_2_1_5_1","unstructured":"\"Performance application programming interface (PAPI) \" http:\/\/icl.cs.utk.edu\/papi\/.  \"Performance application programming interface (PAPI) \" http:\/\/icl.cs.utk.edu\/papi\/."},{"key":"e_1_3_2_1_6_1","unstructured":"\"Gromacs \" http:\/\/www.gromacs.org 2008.  \"Gromacs \" http:\/\/www.gromacs.org 2008."},{"key":"e_1_3_2_1_7_1","volume-title":"Dec","year":"2012","unstructured":"\"Itrs interconnect working group , winter update,\" http:\/\/www.itrs.net\/ , Dec 2012 . \"Itrs interconnect working group, winter update,\" http:\/\/www.itrs.net\/, Dec 2012."},{"key":"e_1_3_2_1_8_1","unstructured":"\"Memory scheduling championship (MSC) \" http:\/\/www.cs.utah.edu\/rajeev\/jwac12\/ 2012.  \"Memory scheduling championship (MSC) \" http:\/\/www.cs.utah.edu\/rajeev\/jwac12\/ 2012."},{"key":"e_1_3_2_1_9_1","unstructured":"\"High bandwidth memory (HBM) dram \" JEDEC JESD235 2013.  \"High bandwidth memory (HBM) dram \" JEDEC JESD235 2013."},{"key":"e_1_3_2_1_10_1","unstructured":"\"Intel 64 and ia-32 architectures software developers \" http:\/\/www.intel.com\/content\/dam\/www\/public\/us\/en\/documents\/manuals\/64-ia-32-architectures-software-developer-vol-3b-part-2-manual.pdf October 2014.  \"Intel 64 and ia-32 architectures software developers \" http:\/\/www.intel.com\/content\/dam\/www\/public\/us\/en\/documents\/manuals\/64-ia-32-architectures-software-developer-vol-3b-part-2-manual.pdf October 2014."},{"key":"e_1_3_2_1_11_1","first-page":"3898","volume-title":"ICASSP 2014","author":"Akin B.","year":"2014","unstructured":"B. Akin , F. Franchetti , and J. C. Hoe , \" FFTS with near-optimal memory access through block data layouts,\" in IEEE International Conference on Acoustics, Speech and Signal Processing , ICASSP 2014 , Florence, Italy, May 4--9 , 2014 , 2014, pp. 3898 -- 3902 . B. Akin, F. Franchetti, and J. C. Hoe, \"FFTS with near-optimal memory access through block data layouts,\" in IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2014, Florence, Italy, May 4--9, 2014, 2014, pp. 3898--3902."},{"key":"e_1_3_2_1_12_1","first-page":"248","volume-title":"ASAP 2014","author":"Akin B.","year":"2014","unstructured":"B. Akin , F. Franchetti , and J. C. Hoe , \" Understanding the design space of dram-optimized hardware FFT accelerators,\" in IEEE 25th International Conference on Application-Specific Systems, Architectures and Processors , ASAP 2014 , Zurich, Switzerland, June 18--20 , 2014 , 2014, pp. 248 -- 255 . B. Akin, F. Franchetti, and J. C. Hoe, \"Understanding the design space of dram-optimized hardware FFT accelerators,\" in IEEE 25th International Conference on Application-Specific Systems, Architectures and Processors, ASAP 2014, Zurich, Switzerland, June 18--20, 2014, 2014, pp. 248--255."},{"key":"e_1_3_2_1_13_1","first-page":"1","volume-title":"HPEC 2014","author":"Akin B.","year":"2014","unstructured":"B. Akin , J. C. Hoe , and F. Franchetti , \" Hamlet: Hardware accelerated memory layout transform within 3d-stacked DRAM,\" in IEEE High Performance Extreme Computing Conference , HPEC 2014 , Waltham, MA, USA, September 9--11 , 2014 , 2014, pp. 1 -- 6 . B. Akin, J. C. Hoe, and F. Franchetti, \"Hamlet: Hardware accelerated memory layout transform within 3d-stacked DRAM,\" in IEEE High Performance Extreme Computing Conference, HPEC 2014, Waltham, MA, USA, September 9--11, 2014, 2014, pp. 1--6."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/FCCM.2012.40"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2485922.2485943"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2004.840311"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/1454115.1454128"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/1583991.1584053"},{"key":"e_1_3_2_1_19_1","first-page":"70","volume-title":"Proceedings. Fifth International Symposium On","author":"Carter J.","year":"1999","unstructured":"J. Carter , W. Hsieh , L. Stoller , M. Swanson , L. Zhang , E. Brunvand , A. Davis , C.-C. Kuo , R. Kuramkote , M. Parker , L. Schaelicke , and T. Tateyama , \" Impulse: building a smarter memory controller,\" in High-Performance Computer Architecture, 1999 . Proceedings. Fifth International Symposium On , Jan 1999 , pp. 70 -- 79 . J. Carter, W. Hsieh, L. Stoller, M. Swanson, L. Zhang, E. Brunvand, A. Davis, C.-C. Kuo, R. Kuramkote, M. Parker, L. Schaelicke, and T. Tateyama, \"Impulse: building a smarter memory controller,\" in High-Performance Computer Architecture, 1999. Proceedings. Fifth International Symposium On, Jan 1999, pp. 70--79."},{"key":"e_1_3_2_1_20_1","volume-title":"Usimm: the utah simulated memory module","author":"Chatterjee N.","year":"2012","unstructured":"N. Chatterjee , R. Balasubramonian , M. Shevgoor , S. Pugsley , A. Udipi , A. Shafiee , K. Sudan , M. Awasthi , and Z. Chishti , \" Usimm: the utah simulated memory module ,\" 2012 . N. Chatterjee, R. Balasubramonian, M. Shevgoor, S. Pugsley, A. Udipi, A. Shafiee, K. Sudan, M. Awasthi, and Z. Chishti, \"Usimm: the utah simulated memory module,\" 2012."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/2063384.2063401"},{"key":"e_1_3_2_1_22_1","first-page":"33","volume-title":"Automation Test in Europe (DATE)","author":"Chen K.","year":"2012","unstructured":"K. Chen , S. Li , N. Muralimanohar , J.-H. Ahn , J. Brockman , and N. Jouppi , \" CACTI-3DD: Architecture-level modeling for 3D die-stacked DRAM main memory,\" in Design , Automation Test in Europe (DATE) , 2012 , pp. 33 -- 38 . K. Chen, S. Li, N. Muralimanohar, J.-H. Ahn, J. Brockman, and N. Jouppi, \"CACTI-3DD: Architecture-level modeling for 3D die-stacked DRAM main memory,\" in Design, Automation Test in Europe (DATE), 2012, pp. 33--38."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSSC.2012.2185184"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2010.50"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/2524713.2524725"},{"key":"e_1_3_2_1_26_1","first-page":"283","volume-title":"2015 IEEE 21st International Symposium on","author":"Farmahini-Farahani A.","year":"2015","unstructured":"A. Farmahini-Farahani , J. H. Ahn , K. Morrow , and N. S. Kim , \" Nda: Near-dram acceleration architecture leveraging commodity dram devices and standard memory modules,\" in High Performance Computer Architecture (HPCA) , 2015 IEEE 21st International Symposium on , Feb 2015 , pp. 283 -- 295 . A. Farmahini-Farahani, J. H. Ahn, K. Morrow, and N. S. Kim, \"Nda: Near-dram acceleration architecture leveraging commodity dram devices and standard memory modules,\" in High Performance Computer Architecture (HPCA), 2015 IEEE 21st International Symposium on, Feb 2015, pp. 283--295."},{"issue":"2","key":"e_1_3_2_1_27_1","first-page":"216","article-title":"The design and implementation of FFTW3","volume":"93","author":"Frigo M.","year":"2005","unstructured":"M. Frigo and S. G. Johnson , \" The design and implementation of FFTW3 ,\" Proceedings of the IEEE, Special issue on \"Program Generation, Optimization, and Platform Adaptation\" , vol. 93 , no. 2 , pp. 216 -- 231 , 2005 . M. Frigo and S. G. Johnson, \"The design and implementation of FFTW3,\" Proceedings of the IEEE, Special issue on \"Program Generation, Optimization, and Platform Adaptation\", vol. 93, no. 2, pp. 216--231, 2005.","journal-title":"Proceedings of the IEEE, Special issue on \"Program Generation, Optimization, and Platform Adaptation\""},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/2.375174"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/1356052.1356053"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/1810085.1810111"},{"key":"e_1_3_2_1_31_1","volume-title":"3d-stacked memory-side acceleration: Accelerator and system design,\" in In the Workshop on Near-Data Processing (WoNDP) (Held in conjunction with MICRO-47.)","author":"Guo Q.","year":"2014","unstructured":"Q. Guo , N. Alachiotis , B. Akin , F. Sadi , G. Xu , T. M. Low , L. Pileggi , J. C. Hoe , and F. Franchetti , \" 3d-stacked memory-side acceleration: Accelerator and system design,\" in In the Workshop on Near-Data Processing (WoNDP) (Held in conjunction with MICRO-47.) , 2014 . Q. Guo, N. Alachiotis, B. Akin, F. Sadi, G. Xu, T. M. Low, L. Pileggi, J. C. Hoe, and F. Franchetti, \"3d-stacked memory-side acceleration: Accelerator and system design,\" in In the Workshop on Near-Data Processing (WoNDP) (Held in conjunction with MICRO-47.), 2014."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/1186736.1186737"},{"key":"e_1_3_2_1_33_1","volume-title":"UCHPC2014","author":"Islam M.","year":"2014","unstructured":"M. Islam , M. Scrback , K. Kavi , M. Ignatowski , and N. Jayasena , \" Improving node-level map-reduce performance using processing-in-memory technologies,\" in 7th Workshop on UnConventional High Performance Computing held in conjunction with the EuroPar 2014, ser . UCHPC2014 , 2014 . M. Islam, M. Scrback, K. Kavi, M. Ignatowski, and N. Jayasena, \"Improving node-level map-reduce performance using processing-in-memory technologies,\" in 7th Workshop on UnConventional High Performance Computing held in conjunction with the EuroPar 2014, ser. UCHPC2014, 2014."},{"key":"e_1_3_2_1_34_1","first-page":"87","volume-title":"2012 Symposium on","author":"Jeddeloh J.","year":"2012","unstructured":"J. Jeddeloh and B. Keeth , \" Hybrid memory cube new dram architecture increases density and performance,\" in VLSI Technology (VLSIT) , 2012 Symposium on , June 2012 , pp. 87 -- 88 . J. Jeddeloh and B. Keeth, \"Hybrid memory cube new dram architecture increases density and performance,\" in VLSI Technology (VLSIT), 2012 Symposium on, June 2012, pp. 87--88."},{"key":"e_1_3_2_1_35_1","first-page":"285","volume-title":"MICRO 31","author":"Kandemir M.","year":"1998","unstructured":"M. Kandemir , A. Choudhary , J. Ramanujam , and P. Banerjee , \" Improving locality using loop and data transformations in an integrated framework,\" in Proceedings of the 31st Annual ACM\/IEEE International Symposium on Microarchitecture, ser . MICRO 31 , 1998 , pp. 285 -- 297 . M. Kandemir, A. Choudhary, J. Ramanujam, and P. Banerjee, \"Improving locality using loop and data transformations in an integrated framework,\" in Proceedings of the 31st Annual ACM\/IEEE International Symposium on Microarchitecture, ser. MICRO 31, 1998, pp. 285--297."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCD.2012.6378608"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2011.89"},{"key":"e_1_3_2_1_38_1","first-page":"56","volume-title":"2013 IEEE International Symposium on","author":"Kestor G.","year":"2013","unstructured":"G. Kestor , R. Gioiosa , D. Kerbyson , and A. Hoisie , \" Quantifying the energy cost of data movement in scientific applications,\" in Workload Characterization (IISWC) , 2013 IEEE International Symposium on , Sept 2013 , pp. 56 -- 65 . G. Kestor, R. Gioiosa, D. Kerbyson, and A. Hoisie, \"Quantifying the energy cost of data movement in scientific applications,\" in Workload Characterization (IISWC), 2013 IEEE International Symposium on, Sept 2013, pp. 56--65."},{"key":"e_1_3_2_1_39_1","first-page":"188","volume-title":"2012 IEEE International","author":"Kim D. H.","year":"2012","unstructured":"D. H. Kim , K. Athikulwongse , M. Healy , M. Hossain , M. Jung , I. Khorosh , G. Kumar , Y.-J. Lee , D. Lewis , T.-W. Lin , C. Liu , S. Panth , M. Pathak , M. Ren , G. Shen , T. Song , D. H. Woo , X. Zhao , J. Kim , H. Choi , G. Loh , H.-H. Lee , and S.-K. Lim , \"3d-maps : 3d massively parallel processor with stacked memory,\" in Solid-State Circuits Conference Digest of Technical Papers (ISSCC) , 2012 IEEE International , Feb 2012 , pp. 188 -- 190 . D. H. Kim, K. Athikulwongse, M. Healy, M. Hossain, M. Jung, I. Khorosh, G. Kumar, Y.-J. Lee, D. Lewis, T.-W. Lin, C. Liu, S. Panth, M. Pathak, M. Ren, G. Shen, T. Song, D. H. Woo, X. Zhao, J. Kim, H. Choi, G. Loh, H.-H. Lee, and S.-K. Lim, \"3d-maps: 3d massively parallel processor with stacked memory,\" in Solid-State Circuits Conference Digest of Technical Papers (ISSCC), 2012 IEEE International, Feb 2012, pp. 188--190."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2008.15"},{"key":"e_1_3_2_1_41_1","first-page":"402","volume-title":"2013 IEEE International. IEEE","author":"Mansuri M.","year":"2013","unstructured":"M. Mansuri , J. E. Jaussi , J. T. Kennedy , T. Hsueh , S. Shekhar , G. Balamurugan , F. O'Mahony , C. Roberts , R. Mooney , and B. Casper , \" A scalable 0.128-to-1tb\/s 0.8-to-2.6 pj\/b 64-lane parallel i\/o in 32nm cmos,\" in Solid-State Circuits Conference Digest of Technical Papers (ISSCC) , 2013 IEEE International. IEEE , 2013 , pp. 402 -- 403 . M. Mansuri, J. E. Jaussi, J. T. Kennedy, T. Hsueh, S. Shekhar, G. Balamurugan, F. O'Mahony, C. Roberts, R. Mooney, and B. Casper, \"A scalable 0.128-to-1tb\/s 0.8-to-2.6 pj\/b 64-lane parallel i\/o in 32nm cmos,\" in Solid-State Circuits Conference Digest of Technical Papers (ISSCC), 2013 IEEE International. IEEE, 2013, pp. 402--403."},{"key":"e_1_3_2_1_42_1","volume-title":"A computer oriented geodetic data base and a new technique in file sequencing","author":"Morton G. M.","year":"1966","unstructured":"G. M. Morton , A computer oriented geodetic data base and a new technique in file sequencing . International Business Machines Company , 1966 . G. M. Morton, A computer oriented geodetic data base and a new technique in file sequencing. International Business Machines Company, 1966."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/279358.279387"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2003.1214317"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/40.592312"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","unstructured":"J. T. Pawlowski \"Hybrid memory cube (HMC) \" in Hotchips 2011.  J. T. Pawlowski \"Hybrid memory cube (HMC) \" in Hotchips 2011.","DOI":"10.1109\/HOTCHIPS.2011.7477494"},{"key":"e_1_3_2_1_47_1","volume-title":"A 0.54 pj\/b 20 gb\/s ground-referenced single-ended short-reach serial link in 28 nm cmos for advanced packaging applications","author":"Poulton J. W.","year":"2013","unstructured":"J. W. Poulton , W. J. Dally , X. Chen , J. G. Eyles , T. H. Greer , S. G. Tell , J. M. Wilson , and C. T. Gray , \" A 0.54 pj\/b 20 gb\/s ground-referenced single-ended short-reach serial link in 28 nm cmos for advanced packaging applications ,\" 2013 . J. W. Poulton, W. J. Dally, X. Chen, J. G. Eyles, T. H. Greer, S. G. Tell, J. M. Wilson, and C. T. Gray, \"A 0.54 pj\/b 20 gb\/s ground-referenced single-ended short-reach serial link in 28 nm cmos for advanced packaging applications,\" 2013."},{"key":"e_1_3_2_1_48_1","volume-title":"Symp. on Perf. Analysis of Sys. and Soft. (ISPASS)","author":"Pugsley S.","year":"2014","unstructured":"S. Pugsley , J. Jestes , H. Zhang , R. Balasubramonian , V. Srinivasan , A. Buyuktosunoglu , A. Davis , and F. Li , \" NDC: Analyzing the impact of 3D-stacked memory+logic devices on mapreduce workloads,\" in Proc. of IEEE Intl . Symp. on Perf. Analysis of Sys. and Soft. (ISPASS) , 2014 . S. Pugsley, J. Jestes, H. Zhang, R. Balasubramonian, V. Srinivasan, A. Buyuktosunoglu, A. Davis, and F. Li, \"NDC: Analyzing the impact of 3D-stacked memory+logic devices on mapreduce workloads,\" in Proc. of IEEE Intl. Symp. on Perf. Analysis of Sys. and Soft. (ISPASS), 2014."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/1502793.1502799"},{"key":"e_1_3_2_1_50_1","volume-title":"SPIRAL: Code generation for DSP transforms,\" Proc. of IEEE, special issue on \"Program Generation, Optimization, and Adaptation","author":"P\u00fcschel M.","unstructured":"M. P\u00fcschel , J. M. F. Moura , J. Johnson , D. Padua , M. Veloso , B. Singer , J. Xiong , F. Franchetti , A. Gacic , Y. Voronenko , K. Chen , R. W. Johnson , and N. Rizzolo , \" SPIRAL: Code generation for DSP transforms,\" Proc. of IEEE, special issue on \"Program Generation, Optimization, and Adaptation \", vol. 93 , no. 2, pp. 232--275, 2005. M. P\u00fcschel, J. M. F. Moura, J. Johnson, D. Padua, M. Veloso, B. Singer, J. Xiong, F. Franchetti, A. Gacic, Y. Voronenko, K. Chen, R. W. Johnson, and N. Rizzolo, \"SPIRAL: Code generation for DSP transforms,\" Proc. of IEEE, special issue on \"Program Generation, Optimization, and Adaptation\", vol. 93, no. 2, pp. 232--275, 2005."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/1995896.1995911"},{"key":"e_1_3_2_1_52_1","volume-title":"Optimizing matrix transpose in CUDA,\" Nvidia CUDA SDK Application Note","author":"Ruetsch G.","year":"2009","unstructured":"G. Ruetsch and P. Micikevicius , \" Optimizing matrix transpose in CUDA,\" Nvidia CUDA SDK Application Note , 2009 . G. Ruetsch and P. Micikevicius, \"Optimizing matrix transpose in CUDA,\" Nvidia CUDA SDK Application Note, 2009."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/2540708.2540725"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/1736020.1736045"},{"key":"e_1_3_2_1_55_1","first-page":"1","volume-title":"May 2012","author":"Sung I.-J.","unstructured":"I.-J. Sung , G. Liu , and W.-M. Hwu , \"Dl : A data layout transformation system for heterogeneous computing,\" in Innovative Parallel Computing (InPar), 2012 , May 2012 , pp. 1 -- 11 . I.-J. Sung, G. Liu, and W.-M. Hwu, \"Dl: A data layout transformation system for heterogeneous computing,\" in Innovative Parallel Computing (InPar), 2012, May 2012, pp. 1--11."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.5555\/130635"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2012.2235125"},{"key":"e_1_3_2_1_58_1","first-page":"1","volume-title":"2010 IEEE 16th International Symposium on. IEEE","author":"Woo D. H.","year":"2010","unstructured":"D. H. Woo , N. H. Seong , D. L. Lewis , and H.-H. Lee , \"An optimized 3d-stacked memory architecture by exploiting excessive, high-density tsv bandwidth,\" in High Performance Computer Architecture (HPCA) , 2010 IEEE 16th International Symposium on. IEEE , 2010 , pp. 1 -- 12 . D. H. Woo, N. H. Seong, D. L. Lewis, and H.-H. Lee, \"An optimized 3d-stacked memory architecture by exploiting excessive, high-density tsv bandwidth,\" in High Performance Computer Architecture (HPCA), 2010 IEEE 16th International Symposium on. IEEE, 2010, pp. 1--12."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/378795.378860"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/2600212.2600213"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/360128.360134"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCD.2005.64"},{"key":"e_1_3_2_1_63_1","first-page":"1","volume-title":"2013 IEEE International","author":"Zhu Q.","year":"2013","unstructured":"Q. Zhu , B. Akin , H. Sumbul , F. Sadi , J. Hoe , L. Pileggi , and F. Franchetti , \" A 3d-stacked logic-in-memory accelerator for application-specific data intensive computing,\" in 3D Systems Integration Conference (3DIC) , 2013 IEEE International , Oct 2013 , pp. 1 -- 7 . Q. Zhu, B. Akin, H. Sumbul, F. Sadi, J. Hoe, L. Pileggi, and F. Franchetti, \"A 3d-stacked logic-in-memory accelerator for application-specific data intensive computing,\" in 3D Systems Integration Conference (3DIC), 2013 IEEE International, Oct 2013, pp. 1--7."}],"event":{"name":"ISCA '15: The 42nd Annual International Symposium on Computer Architecture","location":"Portland Oregon","acronym":"ISCA '15","sponsor":["IEEE TCCA IEEE Computer Society Technical Committee on Computer Architecture","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 42nd Annual International Symposium on Computer Architecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2749469.2750397","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2749469.2750397","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T03:00:40Z","timestamp":1750215640000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2749469.2750397"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,6,13]]},"references-count":63,"alternative-id":["10.1145\/2749469.2750397","10.1145\/2749469"],"URL":"https:\/\/doi.org\/10.1145\/2749469.2750397","relation":{"is-identical-to":[{"id-type":"doi","id":"10.1145\/2872887.2750397","asserted-by":"object"}]},"subject":[],"published":{"date-parts":[[2015,6,13]]},"assertion":[{"value":"2015-06-13","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}