{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T04:54:37Z","timestamp":1764996877074},"publisher-location":"Berlin, Heidelberg","reference-count":29,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642119699"},{"type":"electronic","value":"9783642119705"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2010]]},"DOI":"10.1007\/978-3-642-11970-5_14","type":"book-chapter","created":{"date-parts":[[2010,3,8]],"date-time":"2010-03-08T00:23:33Z","timestamp":1268007813000},"page":"244-263","source":"Crossref","is-referenced-by-count":131,"title":["Automatic C-to-CUDA Code Generation for Affine Programs"],"prefix":"10.1007","author":[{"given":"Muthu Manikandan","family":"Baskaran","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"J.","family":"Ramanujam","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"P.","family":"Sadayappan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"14_CR1","doi-asserted-by":"crossref","unstructured":"Ancourt, C., Irigoin, F.: Scanning polyhedra with do loops. In: PPoPP 1991, pp. 39\u201350 (1991)","DOI":"10.1145\/109625.109631"},{"key":"14_CR2","doi-asserted-by":"crossref","unstructured":"Baskaran, M., Bondhugula, U., Krishnamoorthy, S., Ramanujam, J., Rountev, A., Sadayappan, P.: A Compiler Framework for Optimization of Affine Loop Nests for GPGPUs. In: ACM ICS (June 2008)","DOI":"10.1145\/1375527.1375562"},{"key":"14_CR3","doi-asserted-by":"crossref","unstructured":"Baskaran, M., Bondhugula, U., Krishnamoorthy, S., Ramanujam, J., Rountev, A., Sadayappan, P.: Automatic Data Movement and Computation Mapping for Multi-level Parallel Architectures with Explicitly Managed Memories. In: ACM SIGPLAN PPoPP (February 2008)","DOI":"10.1145\/1345206.1345210"},{"key":"14_CR4","doi-asserted-by":"crossref","unstructured":"Bastoul, C.: Code generation in the polyhedral model is easier than you think. In: PACT 2004, pp. 7\u201316 (2004)","DOI":"10.1109\/PACT.2004.1342537"},{"key":"14_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"132","DOI":"10.1007\/978-3-540-78791-4_9","volume-title":"Compiler Construction","author":"U. Bondhugula","year":"2008","unstructured":"Bondhugula, U., Baskaran, M., Krishnamoorthy, S., Ramanujam, J., Rountev, A., Sadayappan, P.: Automatic transformations for communication-minimized parallelization and locality optimization in the polyhedral model. In: Hendren, L. (ed.) CC 2008. LNCS, vol.\u00a04959, pp. 132\u2013146. Springer, Heidelberg (2008)"},{"key":"14_CR6","doi-asserted-by":"crossref","unstructured":"Bondhugula, U., Hartono, A., Ramanujan, J., Sadayappan, P.: A practical automatic polyhedral parallelizer and locality optimizer. In: ACM SIGPLAN Programming Languages Design and Implementation, PLDI 2008 (2008)","DOI":"10.1145\/1375581.1375595"},{"key":"14_CR7","unstructured":"CLooG: The Chunky Loop Generator, http:\/\/www.cloog.org"},{"key":"14_CR8","doi-asserted-by":"crossref","unstructured":"Fatahalian, K., Sugerman, J., Hanrahan, P.: Understanding the efficiency of GPU algorithms for matrix-matrix multiplication. In: ACM SIGGRAPH\/EUROGRAPHICS Conference on Graphics Hardware, pp. 133\u2013137 (2004)","DOI":"10.1145\/1058129.1058148"},{"issue":"1","key":"14_CR9","first-page":"23","volume":"20","author":"P. Feautrier","year":"1991","unstructured":"Feautrier, P.: Dataflow analysis of array and scalar references. IJPP\u00a020(1), 23\u201353 (1991)","journal-title":"IJPP"},{"issue":"5","key":"14_CR10","first-page":"313","volume":"21","author":"P. Feautrier","year":"1992","unstructured":"Feautrier, P.: Some efficient solutions to the affine scheduling problem, part I: one-dimensional time. IJPP\u00a021(5), 313\u2013348 (1992)","journal-title":"IJPP"},{"key":"14_CR11","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"crossref","first-page":"79","DOI":"10.1007\/3-540-61736-1_44","volume-title":"The Data Parallel Programming Model","author":"P. Feautrier","year":"1996","unstructured":"Feautrier, P.: Automatic parallelization in the polytope model. In: Perrin, G.-R., Darte, A. (eds.) The Data Parallel Programming Model. LNCS, vol.\u00a01132, pp. 79\u2013103. Springer, Heidelberg (1996)"},{"key":"14_CR12","series-title":"Lecture Notes in Computer Science","volume-title":"Software Composition","author":"N.K. Govindaraju","year":"2006","unstructured":"Govindaraju, N.K., Larsen, S., Gray, J., Manocha, D.: A memory model for scientific algorithms on graphics processors. In: L\u00f6we, W., S\u00fcdholt, M. (eds.) SC 2006. LNCS, vol.\u00a04089. Springer, Heidelberg (2006)"},{"key":"14_CR13","unstructured":"General-Purpose Computation Using Graphics Hardware, http:\/\/www.gpgpu.org\/"},{"key":"14_CR14","unstructured":"Griebl, M.: Automatic Parallelization of Loop Programs for Distributed Memory Architectures. Habilitation Thesis. FMI, University of Passau (2004)"},{"key":"14_CR15","doi-asserted-by":"crossref","unstructured":"Irigoin, F., Triolet, R.: Supernode partitioning. In: Proceedings of POPL 1988, pp. 319\u2013329 (1988)","DOI":"10.1145\/73560.73588"},{"key":"14_CR16","unstructured":"Nyland, L., Harris, M., Prins, J.F.: Fast N-body Simulation with CUDA. GPU Gems 3 article (August 2007)"},{"key":"14_CR17","doi-asserted-by":"crossref","unstructured":"Lee, S., Min, S.-J., Eigenmann, R.: Openmp to gpgpu: A compiler framework for automatic translation and optimization. In: PPoPP 2009, pp. 101\u2013110 (2009)","DOI":"10.1145\/1594835.1504194"},{"key":"14_CR18","unstructured":"Lim, A.: Improving Parallelism And Data Locality With Affine Partitioning. PhD thesis, Stanford University (August 2001)"},{"key":"14_CR19","unstructured":"Liu, Y., Zhang, E.Z., Shen, X.: A cross-input adaptive framework for gpu programs optimizations. In: IPDPS (May 2009)"},{"key":"14_CR20","unstructured":"NVIDIA CUDA, http:\/\/developer.nvidia.com\/object\/cuda.html"},{"key":"14_CR21","unstructured":"Parboil Benchmark Suite, http:\/\/impact.crhc.illinois.edu\/parboil.php"},{"key":"14_CR22","unstructured":"Pluto: A polyhedral automatic parallelizer and locality optimizer for multicores http:\/\/pluto-compiler.sourceforge.net"},{"key":"14_CR23","doi-asserted-by":"crossref","unstructured":"Pouchet, L.-N., Bastoul, C., Cohen, A., Vasilache, N.: Iterative optimization in the polyhedral model: Part I, one-dimensional time. In: CGO 2007, pp. 144\u2013156 (2007)","DOI":"10.1109\/CGO.2007.21"},{"key":"14_CR24","doi-asserted-by":"publisher","first-page":"102","DOI":"10.1145\/135226.135233","volume":"8","author":"W. Pugh","year":"1992","unstructured":"Pugh, W.: The Omega test: a fast and practical integer programming algorithm for dependence analysis. Communications of the ACM\u00a08, 102\u2013114 (1992)","journal-title":"Communications of the ACM"},{"issue":"5","key":"14_CR25","first-page":"469","volume":"28","author":"F. Quiller\u00e9","year":"2000","unstructured":"Quiller\u00e9, F., Rajopadhye, S.V., Wilde, D.: Generation of efficient nested loops from polyhedra. IJPP\u00a028(5), 469\u2013498 (2000)","journal-title":"IJPP"},{"key":"14_CR26","doi-asserted-by":"crossref","unstructured":"Ryoo, S., Rodrigues, C., Baghsorkhi, S., Stone, S., Kirk, D., Hwu, W.: Optimization principles and application performance evaluation of a multithreaded GPU using CUDA. In: ACM SIGPLAN PPoPP 2008 (February 2008)","DOI":"10.1145\/1345206.1345220"},{"key":"14_CR27","unstructured":"Ryoo, S., Rodrigues, C., Stone, S., Baghsorkhi, S., Ueng, S., Hwu, W.: Program optimization study on a 128-core GPU. In: The First Workshop on General Purpose Processing on Graphics Processing Units (October 2007)"},{"key":"14_CR28","doi-asserted-by":"crossref","unstructured":"Ryoo, S., Rodrigues, C., Stone, S., Baghsorkhi, S., Ueng, S., Stratton, J., Hwu, W.: Program optimization space pruning for a multithreaded GPU. In: CGO (2008)","DOI":"10.1145\/1356058.1356084"},{"key":"14_CR29","doi-asserted-by":"crossref","unstructured":"Vasilache, N., Bastoul, C., Girbal, S., Cohen, A.: Violated dependence analysis. In: ACM ICS (June 2006)","DOI":"10.1145\/1183401.1183448"}],"container-title":["Lecture Notes in Computer Science","Compiler Construction"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-11970-5_14.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,11,24]],"date-time":"2020-11-24T02:45:57Z","timestamp":1606185957000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-11970-5_14"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2010]]},"ISBN":["9783642119699","9783642119705"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-11970-5_14","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2010]]}}}