{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2023,1,13]],"date-time":"2023-01-13T12:48:57Z","timestamp":1673614137902},"reference-count":30,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2015,6,23]],"date-time":"2015-06-23T00:00:00Z","timestamp":1435017600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Sign Process Syst"],"published-print":{"date-parts":[[2016,10]]},"DOI":"10.1007\/s11265-015-1018-0","type":"journal-article","created":{"date-parts":[[2015,6,22]],"date-time":"2015-06-22T03:18:47Z","timestamp":1434943127000},"page":"67-82","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["FFTs with Near-Optimal Memory Access Through Block Data Layouts: Algorithm, Architecture and Design Automation"],"prefix":"10.1007","volume":"85","author":[{"given":"Berkin","family":"Akin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Franz","family":"Franchetti","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"James C.","family":"Hoe","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,6,23]]},"reference":[{"key":"1018_CR1","unstructured":"CACTI 6.5. HP labs. http:\/\/www.hpl.hp.com\/research\/cacti\/ ."},{"key":"1018_CR2","unstructured":"DDR3-1600 dram datasheet. MT41J256M4, Micron. http:\/\/www.micron.com\/parts\/dram\/ddr3-sdram ."},{"key":"1018_CR3","unstructured":"DDR3 sdram system-power calculator. Micron. http:\/\/www.micron.com\/products\/support\/power-calc ."},{"key":"1018_CR4","unstructured":"DesignWare library. Synopsys. http:\/\/www.synopsys.com\/dw ."},{"key":"1018_CR5","unstructured":"McPAT 1.0. HP labs. http:\/\/www.hpl.hp.com\/research\/mcpat\/ ."},{"key":"1018_CR6","unstructured":"CUDA toolkit 5.0 performance report (2013). Nvidia. https:\/\/developer.nvidia.com\/cuda-math-library ."},{"key":"1018_CR7","doi-asserted-by":"crossref","unstructured":"Akin, B., Franchetti, F., & Hoe, J.C. (2014). FFTs with near-optimal memory access through block data layouts. In Proceedings of IEEE international conference on acoustics speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP.2014.6854332"},{"key":"1018_CR8","doi-asserted-by":"crossref","unstructured":"Akin, B., Franchetti, F., & Hoe, J.C. (2014). Understanding the design space of dram-optimized hardware FFT accelerators. In IEEE 25th international conference on application-specific systems, architectures and processors, ASAP 2014, June 18-20, (pp. 248\u2013255). Zurich.","DOI":"10.1109\/ASAP.2014.6868669"},{"key":"1018_CR9","doi-asserted-by":"crossref","unstructured":"Akin, B., Franchetti, F., & Hoe, J.C. (2015). Data reorganization in memory using 3d-stacked dram. In Proceedings of the 42nd international symposium on computer architecture (ISCA).","DOI":"10.1145\/2749469.2750397"},{"key":"1018_CR10","doi-asserted-by":"crossref","unstructured":"Akin, B., Hoe, J.C., & Franchetti, F. (2014). Hamlet: hardware accelerated memory layout transform within 3d-stacked DRAM. In IEEE high performance extreme computing conference, HPEC 2014, September 9-11 (pp. 1\u20136). Waltham.","DOI":"10.1109\/HPEC.2014.7040954"},{"key":"1018_CR11","doi-asserted-by":"crossref","unstructured":"Akin, B., Milder, P.A., Franchetti, F., & Hoe, J.C. (2012). Memory bandwidth efficient two-dimensional fast Fourier transform algorithm and implementation for large problem sizes. In Proceedings of the IEEE symposium on FCCM (pp. 188\u2013191).","DOI":"10.1109\/FCCM.2012.40"},{"key":"1018_CR12","unstructured":"Chen, K., Li, S., Muralimanohar, N., Ahn, J.H., Brockman, J., & Jouppi, N. (2012). CACTI-3DD: architecture-level modeling for 3D die-stacked DRAM main memory. In Design, automation test in Europe (DATE) (pp. 33\u201338)."},{"key":"1018_CR13","doi-asserted-by":"crossref","unstructured":"Chung, E.S., Milder, P.A., Hoe, J.C., & Mai, K. (2010). Single-chip heterogeneous computing: does the future include custom logic, FPGAs, and GPGPUs?. In Proceedings of the 43th IEEE\/ACM international symposium on microarchitecture (MICRO).","DOI":"10.1109\/MICRO.2010.36"},{"issue":"90","key":"1018_CR14","doi-asserted-by":"crossref","first-page":"297","DOI":"10.1090\/S0025-5718-1965-0178586-1","volume":"19","author":"JW Cooley","year":"1965","unstructured":"Cooley, J.W., & Tukey, J.W. (1965). An algorithm for the machine calculation of complex Fourier series. Mathematics of Computation, 19(90), 297\u2013301.","journal-title":"Mathematics of Computation"},{"issue":"2.3","key":"1018_CR15","doi-asserted-by":"crossref","first-page":"457","DOI":"10.1147\/rd.492.0457","volume":"49","author":"M Eleftheriou","year":"2005","unstructured":"Eleftheriou, M., Fitch, B., Rayshubskiy, A., Ward, T.J.C., & Germain, R. (2005). Scalable framework for 3D FFTs on the Blue Gene\/L supercomputer: implementation and early performance measurements. IBM Journal of Research and Development, 49(2.3), 457\u2013464.","journal-title":"IBM Journal of Research and Development"},{"key":"1018_CR16","doi-asserted-by":"crossref","unstructured":"Franchetti, F., & P\u00fcschel, M. (2011). Encyclopedia of Parallel Computing, chap, Fast fourier transform: Springer.","DOI":"10.1007\/978-0-387-09766-4_243"},{"issue":"2","key":"1018_CR17","first-page":"216","volume":"93","author":"M Frigo","year":"2005","unstructured":"Frigo, M., & Johnson, S.G. (2005). The design and implementation of FFTW3. Proceedings of the IEEE, Special issue on Program Generation Optimization, and Platform Adaptation, 93(2), 216\u2013231.","journal-title":"Proceedings of the IEEE, Special issue on Program Generation Optimization, and Platform Adaptation"},{"key":"1018_CR18","doi-asserted-by":"crossref","unstructured":"Govindaraju, N.K., Lloyd, B., Dotsenko, Y., Smith, B., & Manferdelli, J. (2008). High performance discrete Fourier transforms on graphics processors. In Proceedings of the ACM\/IEEE conference on supercomputing (SC) (pp. 2:1\u20132:12).","DOI":"10.1109\/SC.2008.5213922"},{"key":"1018_CR19","doi-asserted-by":"crossref","unstructured":"Loh, G.H. (2008). 3D-stacked memory architectures for multi-core processors. In Proceedings of the 35th annual international symposium on computer architecture, (ISCA) (pp. 453\u2013464).","DOI":"10.1145\/1394608.1382159"},{"key":"1018_CR20","doi-asserted-by":"crossref","unstructured":"Milder, P.A., Franchetti, F., Hoe, J.C., & P\u00fcschel, M. (2012). Computer generation of hardware for linear digital signal processing transforms. ACM Transactions on Design Automation of Electronic Systems, 17(2).","DOI":"10.1145\/2159542.2159547"},{"key":"1018_CR21","doi-asserted-by":"crossref","unstructured":"Milder, P.A., Hoe, J.C., & P\u00fcschel, M. (2009). Automatic generation of streaming datapaths for arbitrary fixed permutations. In Design, automation and test in Europe (DATE) (pp. 1118\u20131123).","DOI":"10.1109\/DATE.2009.5090831"},{"key":"1018_CR22","doi-asserted-by":"crossref","unstructured":"Pawlowski, J.T. (2011). Hybrid memory cube (HMC). In Hotchips.","DOI":"10.1109\/HOTCHIPS.2011.7477494"},{"key":"1018_CR23","doi-asserted-by":"crossref","unstructured":"Pedram, A., McCalpin, J., & Gerstlauer, A. (2013). Transforming a linear algebra core to an FFT accelerator. In Proceedings of IEEE international conference on application-specific systems, architectures and processors (ASAP) (pp. 175\u2013184).","DOI":"10.1109\/ASAP.2013.6567572"},{"issue":"2","key":"1018_CR24","first-page":"232","volume":"93","author":"M P\u00fcschel","year":"2005","unstructured":"P\u00fcschel, M., Moura, J.M.F., Johnson, J., Padua, D., Veloso, M., Singer, B., Xiong, J., Franchetti, F., Gacic, A., Voronenko, Y., Chen, K., Johnson, R.W., & Rizzolo, N. (2005). SPIRAL: code generation for DSP transforms. Proceedings of IEEE, Special Issue on Program Generation Optimization, and Adaptation, 93(2), 232\u2013275.","journal-title":"Proceedings of IEEE, Special Issue on Program Generation Optimization, and Adaptation"},{"issue":"1","key":"1018_CR25","doi-asserted-by":"crossref","first-page":"16","DOI":"10.1109\/L-CA.2011.4","volume":"10","author":"P Rosenfeld","year":"2011","unstructured":"Rosenfeld, P., Cooper-Balis, E., & Jacob, B. (2011). Dramsim2: a cycle accurate memory system simulator. IEEE Computer Architecture Letters, 10(1), 16\u201319.","journal-title":"IEEE Computer Architecture Letters"},{"key":"1018_CR26","doi-asserted-by":"crossref","unstructured":"Loan Van, C. (1992). Computational frameworks for the fast Fourier transform: SIAM.","DOI":"10.1137\/1.9781611970999"},{"issue":"4","key":"1018_CR27","doi-asserted-by":"crossref","first-page":"597","DOI":"10.1109\/TCAD.2012.2235125","volume":"32","author":"C Weis","year":"2013","unstructured":"Weis, C., Loi, I., Benini, L., & Wehn, N. (2013). Exploration and optimization of 3-D integrated dram subsystems. IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems, 32(4), 597\u2013610.","journal-title":"IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems"},{"issue":"4","key":"1018_CR28","first-page":"755","volume":"58","author":"CL Yu","year":"2010","unstructured":"Yu, C.L., Irick, K., Chakrabarti, C., & Narayanan, V. (2010). Multidimensional DFT IP generator for FPGA platforms. IEEE Transactions on Circuits and Systems, 58(4), 755\u2013764.","journal-title":"IEEE Transactions on Circuits and Systems"},{"key":"1018_CR29","doi-asserted-by":"crossref","unstructured":"Zhu, Q., Akin, B., Sumbul, H., Sadi, F., Hoe, J., Pileggi, L., & Franchetti, F. (2013). A 3D-stacked logic-in-memory accelerator for application-specific data intensive computing. In 2013 IEEE international 3D systems integration conference (3DIC) (pp. 1\u20137).","DOI":"10.1109\/3DIC.2013.6702348"},{"key":"1018_CR30","doi-asserted-by":"crossref","unstructured":"Zhu, Q., Vaidyanathan, K., Shacham, O., Horowitz, M., Pileggi, L., & Franchetti, F. (2012). Design automation framework for application-specific logic-in-memory blocks. In Proceedings of IEEE international conference on application-specific systems, architectures and processors (ASAP) (pp. 125\u2013132).","DOI":"10.1109\/ASAP.2012.21"}],"container-title":["Journal of Signal Processing Systems"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-015-1018-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11265-015-1018-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-015-1018-0","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,8,27]],"date-time":"2019-08-27T09:43:32Z","timestamp":1566899012000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11265-015-1018-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,6,23]]},"references-count":30,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2016,10]]}},"alternative-id":["1018"],"URL":"https:\/\/doi.org\/10.1007\/s11265-015-1018-0","relation":{},"ISSN":["1939-8018","1939-8115"],"issn-type":[{"value":"1939-8018","type":"print"},{"value":"1939-8115","type":"electronic"}],"subject":[],"published":{"date-parts":[[2015,6,23]]}}}