{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2023,1,9]],"date-time":"2023-01-09T05:26:09Z","timestamp":1673241969686},"reference-count":39,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2008,3,16]],"date-time":"2008-03-16T00:00:00Z","timestamp":1205625600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2009,2]]},"DOI":"10.1007\/s11227-008-0186-0","type":"journal-article","created":{"date-parts":[[2008,3,15]],"date-time":"2008-03-15T15:12:35Z","timestamp":1205593955000},"page":"171-197","source":"Crossref","is-referenced-by-count":5,"title":["Matrix-based streamization approach for improving locality and parallelism on FT64 stream processor"],"prefix":"10.1007","volume":"47","author":[{"given":"Xuejun","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jing","family":"Du","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaobo","family":"Yan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Deng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2008,3,16]]},"reference":[{"key":"186_CR1","doi-asserted-by":"crossref","unstructured":"Kapasi UJ, Rixner S, Dally WJ et al (2003) Programmable stream processors. IEEE Comput 54\u201362","DOI":"10.1109\/MC.2003.1220582"},{"key":"186_CR2","unstructured":"Khailany B (2003) The VLSI implementation and evaluation of area-and energy-efficient streaming media processors. Ph.D. thesis, Stanford University"},{"issue":"2","key":"186_CR3","doi-asserted-by":"crossref","first-page":"25","DOI":"10.1109\/MM.2002.997877","volume":"22","author":"M Taylor","year":"2002","unstructured":"Taylor M, Kim J, Miller J et al. (2002) The RAW microprocessor: a computational fabric for software circuits and general purpose programs. IEEE Micro 22(2):25\u201335","journal-title":"IEEE Micro"},{"issue":"7","key":"186_CR4","doi-asserted-by":"crossref","first-page":"44","DOI":"10.1109\/MC.2004.65","volume":"37","author":"D Burger","year":"2004","unstructured":"Burger D, Keckler SW, McKinley KS et al. (2004) Scaling to the end of silicon with EDGE architectures. Computer 37(7):44\u201355","journal-title":"Computer"},{"key":"186_CR5","doi-asserted-by":"crossref","unstructured":"Gordon MI, Thies W, Amarasinghe S (2006) Exploiting coarse-grained task, data, and pipeline parallelism in stream programs. In: Proceedings of ASPLOS\u201906, California, USA","DOI":"10.1145\/1168857.1168877"},{"key":"186_CR6","unstructured":"Andrew AL, Thies W, Amarasinghe S (2003) Linear analysis and optimization of stream programs. In: Proceedings of the SIGPLAN\u201903 conference on programming language design and implementation, San Diego, CA"},{"key":"186_CR7","doi-asserted-by":"crossref","unstructured":"Owens JD, Rixner S et al (2002) Media processing applications on the imagine stream processor. In: Proceedings of the 2002 international conference on computer design","DOI":"10.1109\/ICCD.2002.1106785"},{"key":"186_CR8","doi-asserted-by":"crossref","first-page":"210","DOI":"10.1145\/1250662.1250689","volume-title":"ISCA\u201907: Proceedings of the 34th annual international symposium on computer architecture","author":"X Yang","year":"2007","unstructured":"Yang X, Yan X, Xing Z et al. (2007) A 64-bit stream processor architecture for scientific applications. In: ISCA\u201907: Proceedings of the 34th annual international symposium on computer architecture. ACM Press, New York, pp 210\u2013219"},{"key":"186_CR9","unstructured":"Amarasinghe S et al (2003) Stream languages and programming models. In: Proceedings of the international conference on parallel architectures and compilation techniques 2003"},{"key":"186_CR10","unstructured":"Mattson P (2002) A\u00a0programming system for the imagine media processor. Ph.D. thesis, Dept of Electrical Engineering, Stanford University"},{"key":"186_CR11","doi-asserted-by":"crossref","unstructured":"Du J, Yang X et al (2007) Architecture-based optimization for mapping scientific applications to imagine. In: ISPA\u201907: Proceedings of the 2007 international symposium on parallel and distributed processing with applications, Ontario, Canada","DOI":"10.1007\/978-3-540-74742-0_6"},{"key":"186_CR12","doi-asserted-by":"crossref","first-page":"33","DOI":"10.1145\/1152154.1152164","volume-title":"PACT\u201906: Proceedings of the 15th international conference on parallel architectures and compilation techniques","author":"A Das","year":"2006","unstructured":"Das A, Dally WJ, Mattson P (2006) Compiling for stream processing. In: PACT\u201906: Proceedings of the 15th international conference on parallel architectures and compilation techniques. ACM Press, New York, pp 33\u201342"},{"key":"186_CR13","unstructured":"Johnsson O, Stenemo M, ul-Abdin Z (2005) Programming & implementation of streaming applications. Master\u2019s thesis, Computer and Electrical Engineering Halmstad University"},{"key":"186_CR14","unstructured":"Ahn JH, Dally WJ et al (2004). Evaluating the imagine stream architecture. In: Proceedings of the annual international symposium on computer architecture 2004"},{"key":"186_CR15","unstructured":"Jayasena NS (2005) Memory hierarchy design for stream computing. Ph.D. thesis, Stanford University"},{"issue":"4","key":"186_CR16","doi-asserted-by":"crossref","first-page":"452","DOI":"10.1109\/71.97902","volume":"2","author":"ME Wolf","year":"1991","unstructured":"Wolf ME, Lam M (1991) A loop transformation theory and an algorithm to maximize parallelism. IEEE Trans Parallel Distrib Syst 2(4):452\u2013471","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"186_CR17","doi-asserted-by":"crossref","unstructured":"Kuck D, Kuhn R et al (1981) Dependence graphs and compiler optimizations. In: Conference record of the eighth annual ACM symposium on the principles of programming languages, Williamsburg, VA, January 1981","DOI":"10.1145\/567532.567555"},{"key":"186_CR18","volume-title":"High performance compilers for parallel computing","author":"MJ Wolfe","year":"1996","unstructured":"Wolfe MJ (1996) High performance compilers for parallel computing. Addison-Wesley, Reading"},{"key":"186_CR19","doi-asserted-by":"crossref","unstructured":"Du J, Yang X et al (2006) Scientific computing applications on the imagine stream processor. In: Proceedings of the 11th Asia-pacific computer systems architecture conference, Shanghai, China","DOI":"10.1007\/11859802_5"},{"key":"186_CR20","unstructured":"Fan Z, Qiu F et al (2004) Gpu cluster for high performance computing. In: Proceedings of supercomputing conference 2004"},{"key":"186_CR21","unstructured":"Harris MJ, Baxter WV et al (2003) Simulation of cloud dynamics on graphics hardware. In: Proceedings of the ACM SIGGRAPH\/EUROGRAPHICS conference on graphics hardware, Switzerland, pp\u00a092\u2013101"},{"issue":"3","key":"186_CR22","doi-asserted-by":"crossref","first-page":"917","DOI":"10.1145\/882262.882364","volume":"22","author":"J Bolz","year":"2003","unstructured":"Bolz J, Farmer I, Grinspun E, Schr \u00d6der P (2003) Sparse matrix solvers on the Gpu: conjugate gradients and multigrid. ACM Trans Graph 22(3):917\u2013924","journal-title":"ACM Trans Graph"},{"key":"186_CR23","unstructured":"Dally WJ, Hanrahan P et al (2003) Merrimac: supercomputing with streams. In: Proceedings of supercomputing conference 2003"},{"key":"186_CR24","doi-asserted-by":"crossref","unstructured":"Erez M, Ahn J et al (2004) Analysis and performance results of a molecular modeling application on Merrimac. In: Proceedings of supercomputing conference 2004","DOI":"10.1109\/SC.2004.69"},{"key":"186_CR25","unstructured":"Erez M (2007) Merrimac\u2014high-performance, highly-efficient scientific computing with streams. Ph.D. thesis, Dept of Electrical Engineering, Stanford University"},{"key":"186_CR26","doi-asserted-by":"crossref","unstructured":"Erez M, Ahn J et al (2007) Executing irregular scientific applications on stream architectures. In: (ICS\u201907): Proceedings of the 21th ACM international conference on supercomputing","DOI":"10.1145\/1274971.1274987"},{"key":"186_CR27","unstructured":"Griem G, Oliker L (2003) Transitive closure on the imagine stream processor. In: Proceedings of the 5th workshop on media and streaming processors, San Diego, CA"},{"key":"186_CR28","doi-asserted-by":"crossref","unstructured":"Ahn J, Dally WJ, Erez M (2007) Tradeoff between data-, instruction-, and thread-level parallelism in stream processors. In: (ICS\u201907): Proceedings of the 21th ACM international conference on supercomputing","DOI":"10.1145\/1274971.1274991"},{"key":"186_CR29","doi-asserted-by":"crossref","unstructured":"Sermulins J, Thies W et al (2005) Cache aware optimization of stream programs. In: Proceedings of LCTES\u201905, Chicago, Illinois, USA","DOI":"10.1145\/1065910.1065927"},{"key":"186_CR30","doi-asserted-by":"crossref","unstructured":"Wolf M, Lam M (1991) A data locality optimizing algorithm. In: Proceedings of ACM SIGPLAN\u201991 conference on programming language design and implementation, Ontario, Canada, pp 30\u201344","DOI":"10.1145\/113445.113449"},{"key":"186_CR31","doi-asserted-by":"crossref","unstructured":"McKinley K, Carr S, Tseng CW (1996) Improving data locality with loop transformations. ACM Trans Program Lang Syst","DOI":"10.1145\/233561.233564"},{"key":"186_CR32","unstructured":"Li W (1993) Compiling for NUMA parallel machines. Ph.D. thesis, Cornell University"},{"issue":"2","key":"186_CR33","doi-asserted-by":"crossref","first-page":"115","DOI":"10.1109\/71.752779","volume":"10","author":"M Kandemir","year":"1999","unstructured":"Kandemir M, Choudhary A et al. (1999) A linear algebra framework for automatic determination of optimal data layouts. IEEE Trans Parallel Distrib Syst 10(2):115\u2013135","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"186_CR34","doi-asserted-by":"crossref","unstructured":"Cierniak M, Li W (1995) Unifying Data and control transformations for distributed shared memory machines. In: ACM SIGPLAN IPDPS, pp 205\u2013217","DOI":"10.1145\/207110.207145"},{"key":"186_CR35","doi-asserted-by":"crossref","unstructured":"Kandemir M, Choudhary A et al (1998) Improving locality using loop and data transformations in an integrated framework. In: Proceedings of international symposium on microarchitecture, pp 285\u2013297","DOI":"10.1109\/MICRO.1998.742790"},{"issue":"9","key":"186_CR36","doi-asserted-by":"crossref","first-page":"922","DOI":"10.1109\/TPDS.2001.1184186","volume":"12","author":"M Kandemir","year":"2001","unstructured":"Kandemir M, Banerjee P et al. (2001) Static and dynamic locality optimizations using integer linear programming. IEEE Trans Parallel Distrib Syst 12(9):922\u2013940","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"186_CR37","doi-asserted-by":"crossref","unstructured":"Kandemir M et al (1999) A graph based framework to detect optimal memory layouts for improving data locality. In: Proceedings of the 13th international parallel processing symposium, San Juan, Puerto Rico, pp 738\u2013743","DOI":"10.1109\/IPPS.1999.760558"},{"key":"186_CR38","unstructured":"O\u2019Boyle M, Knijnenburg P (1996) Non-singular data transformations: definition, validity, applications. In: Proceedings of 6th workshop on compilers for parallel computers, pp 287\u2013297"},{"key":"186_CR39","doi-asserted-by":"crossref","unstructured":"Garcia J, Ayguade E et al (1996) Dynamic data distribution with control flow analysis. In: Proceedings of supercomputing conference 1996","DOI":"10.1145\/369028.369048"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-008-0186-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11227-008-0186-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-008-0186-0","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,9,5]],"date-time":"2021-09-05T23:49:55Z","timestamp":1630885795000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11227-008-0186-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2008,3,16]]},"references-count":39,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2009,2]]}},"alternative-id":["186"],"URL":"https:\/\/doi.org\/10.1007\/s11227-008-0186-0","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2008,3,16]]}}}