{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T16:01:53Z","timestamp":1780675313250,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":75,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,9,30]],"date-time":"2020-09-30T00:00:00Z","timestamp":1601424000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the National Natural Science Foundation of China","award":["61802368, 61521092, 61432016, 61432018, 61332009, 61702485, 61872043"],"award-info":[{"award-number":["61802368, 61521092, 61432016, 61432018, 61332009, 61702485, 61872043"]}]},{"name":"the National Key Research and Development Program of China","award":["2017YFB0202002"],"award-info":[{"award-number":["2017YFB0202002"]}]},{"name":"CCF-Tencent Open Research Fund"},{"name":"Australian Research Council grant","award":["DP170103956, DP180104069"],"award-info":[{"award-number":["DP170103956, DP180104069"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,9,30]]},"DOI":"10.1145\/3410463.3414637","type":"proceedings-article","created":{"date-parts":[[2020,9,30]],"date-time":"2020-09-30T10:43:04Z","timestamp":1601462584000},"page":"97-109","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":10,"title":["Bandwidth-Aware Loop Tiling for DMA-Supported Scratchpad Memory"],"prefix":"10.1145","author":[{"given":"Mingchuan","family":"Wu","sequence":"first","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences &amp; University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ying","family":"Liu","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huimin","family":"Cui","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences &amp; University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingfu","family":"Wei","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences &amp; University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Quanfeng","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences &amp; University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Limin","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fang","family":"Lv","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingling","family":"Xue","sequence":"additional","affiliation":[{"name":"University of New South Wales, Sydney, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaobing","family":"Feng","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences &amp; University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2020,9,30]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2014. LLVM-CBE. https:\/\/github.com\/JuliaComputing\/llvm-cbe  2014. LLVM-CBE. https:\/\/github.com\/JuliaComputing\/llvm-cbe"},{"key":"e_1_3_2_1_2_1","unstructured":"A. V. Aho M. S. Lam R. Sethi and J. D. Ullman. 2011. Compilers Principles Techniques and Tools (2 ed.).  A. V. Aho M. S. Lam R. Sethi and J. D. Ullman. 2011. Compilers Principles Techniques and Tools (2 ed.)."},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of the 31th IEEE International Parallel and Distributed Processing Symposium (IPDPS '17)","author":"Ao Y.","unstructured":"Y. Ao , C. Yang , X. Wang , W. Xue , H. Fu , F. Liu , L. Gan , P. Xu , and W. Ma . 2017. 26 PFLOPS Stencil Computations for Atmospheric Modeling on Sunway TaihuLight . In Proceedings of the 31th IEEE International Parallel and Distributed Processing Symposium (IPDPS '17) . IEEE, Florida USA. Y. Ao, C. Yang, X. Wang, W. Xue, H. Fu, F. Liu, L. Gan, P. Xu, and W. Ma. 2017. 26 PFLOPS Stencil Computations for Atmospheric Modeling on Sunway TaihuLight. In Proceedings of the 31th IEEE International Parallel and Distributed Processing Symposium (IPDPS '17). IEEE, Florida USA."},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the 10th International Symposium on Hardware\/Software Codesign (CODES '02)","author":"Banakar R.","unstructured":"R. Banakar , S. Steinke , B. Lee , M. Balakrishnan , and P. Marwedel . 2002. Scratchpad memory: a design alternative for cache on-chip memory in embedded systems . In Proceedings of the 10th International Symposium on Hardware\/Software Codesign (CODES '02) . New York, NY, USA, 73--78. R. Banakar, S. Steinke, B. Lee, M. Balakrishnan, and P. Marwedel. 2002. Scratchpad memory: a design alternative for cache on-chip memory in embedded systems. In Proceedings of the 10th International Symposium on Hardware\/Software Codesign (CODES '02). New York, NY, USA, 73--78."},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of the 8th Annual IEEE\/ACM International Symposium on Code Generation and Optimization (CGO '10)","author":"Baskaran M.","unstructured":"M. Baskaran , A. Hartono , S. Tavarageri , T. Henretty , J. Ramanujam , and P. Sadayappan . 2010. Parameterized Tiling Revisited . In Proceedings of the 8th Annual IEEE\/ACM International Symposium on Code Generation and Optimization (CGO '10) . ACM, New York, NY, USA, 200--209. M. Baskaran, A. Hartono, S. Tavarageri, T. Henretty, J.Ramanujam, and P. Sadayappan. 2010. Parameterized Tiling Revisited. In Proceedings of the 8th Annual IEEE\/ACM International Symposium on Code Generation and Optimization (CGO '10). ACM, New York, NY, USA, 200--209."},{"key":"e_1_3_2_1_6_1","volume-title":"Proceedings of the 15th International Symposium on High-Performance Computer Architecture (HPCA '09)","author":"Bhatotia P. K.","unstructured":"P. K. Bhatotia , S. K. Aggarwal , and M. Chaudhuri . 2009. A Compilation Framework for Irregular Memory Accesses on the Cell Broadband Engine . In Proceedings of the 15th International Symposium on High-Performance Computer Architecture (HPCA '09) . IEEE, North Carolina, USA. P. K. Bhatotia, S. K. Aggarwal, and M. Chaudhuri. 2009. A Compilation Framework for Irregular Memory Accesses on the Cell Broadband Engine. In Proceedings of the 15th International Symposium on High-Performance Computer Architecture (HPCA '09). IEEE, North Carolina, USA."},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the Conference on Design, Automation and Test in Europe (DATE '06)","author":"Chen G.","unstructured":"G. Chen , O. Ozturk , M. T. Kandemir , and M. Karak\u00f6y . 2006. Dynamic Scratch-Pad Memory Management for Irregular Array Access Patterns . In Proceedings of the Conference on Design, Automation and Test in Europe (DATE '06) . Munich, Germany. G. Chen, O. Ozturk, M. T. Kandemir, and M. Karak\u00f6y. 2006. Dynamic Scratch-Pad Memory Management for Irregular Array Access Patterns. In Proceedings of the Conference on Design, Automation and Test in Europe (DATE '06). Munich, Germany."},{"key":"e_1_3_2_1_8_1","unstructured":"J. Chen R. Tan and Y. Zhang. 2017. Heterogeneous Parallel and Distributed Optimization of K-means Algorithm on Sunway Supercomputer. In Proceedings of the 15th IEEE International Symposium on Parallel and Distributed Processing with Applications and the 16th IEEE International Conference on Ubiquitous Computing and Communications (ISPA '17). IEEE Guangzhou China.  J. Chen R. Tan and Y. Zhang. 2017. Heterogeneous Parallel and Distributed Optimization of K-means Algorithm on Sunway Supercomputer. In Proceedings of the 15th IEEE International Symposium on Parallel and Distributed Processing with Applications and the 16th IEEE International Conference on Ubiquitous Computing and Communications (ISPA '17). IEEE Guangzhou China."},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of the 19th International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS '14)","author":"Chen T.","unstructured":"T. Chen , Z. Du , J. Wang , C. Wu , and Y. Chen . 2014. DianNao: A Small-Footprint High-Throughput Accelerator for Ubiquitous Machine-Learning . In Proceedings of the 19th International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS '14) . ACM, Salt Lake City, Utah, USA. T. Chen, Z. Du, J. Wang, C. Wu, and Y. Chen. 2014. DianNao: A Small-Footprint High-Throughput Accelerator for Ubiquitous Machine-Learning. In Proceedings of the 19th International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS '14). ACM, Salt Lake City, Utah, USA."},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the 2007 International Conference on Compilers, Architecture, and Synthesis for Embedded Systems (CASES'07)","author":"Cho D.","unstructured":"D. Cho , I. Issenin , N. Dutt , J. W. Yoon , and Y. Paek . 2007. Software Controlled Memory Layout Reorganization for Irregular Array Access Patterns . In Proceedings of the 2007 International Conference on Compilers, Architecture, and Synthesis for Embedded Systems (CASES'07) . Salzburg, Austria. D. Cho, I. Issenin, N. Dutt, J. W. Yoon, and Y. Paek. 2007. Software Controlled Memory Layout Reorganization for Irregular Array Access Patterns. In Proceedings of the 2007 International Conference on Compilers, Architecture, and Synthesis for Embedded Systems (CASES'07). Salzburg, Austria."},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the 2011 IEEE International Parallel and Distributed Processing Symposium (IPDPS '11)","author":"Christen M.","unstructured":"M. Christen , O. Schenk , and H. Burkhart . 2011. PATUS: A code generation and autotuning framework for parallel iterative stencil computations on modern microarchitectures . In Proceedings of the 2011 IEEE International Parallel and Distributed Processing Symposium (IPDPS '11) . IEEE, New Orleans, Louisiana, USA, 676--687. M. Christen, O. Schenk, and H. Burkhart. 2011. PATUS: A code generation and autotuning framework for parallel iterative stencil computations on modern microarchitectures. In Proceedings of the 2011 IEEE International Parallel and Distributed Processing Symposium (IPDPS '11). IEEE, New Orleans, Louisiana, USA, 676--687."},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the ACM SIGPLAN 1995 Conference on Programming Language Design and Implementation (PLDI '95)","author":"Coleman S.","unstructured":"S. Coleman and K. S. McKinley . 1995. Tile Size Selection Using Cache Organization and Data Layout . In Proceedings of the ACM SIGPLAN 1995 Conference on Programming Language Design and Implementation (PLDI '95) . ACM, New York, NY, USA, 279--290. S. Coleman and K. S. McKinley. 1995. Tile Size Selection Using Cache Organization and Data Layout. In Proceedings of the ACM SIGPLAN 1995 Conference on Programming Language Design and Implementation (PLDI '95). ACM, New York, NY, USA, 279--290."},{"key":"e_1_3_2_1_13_1","volume-title":"2011 IEEE International Parallel Distributed Processing Symposium. IEEE, 255--265","author":"Cui H.","unstructured":"H. Cui , L. Wang , Y J. Xue , Yang, and X. Feng . 2011. Automatic Library Generation for BLAS3 on GPUs . In 2011 IEEE International Parallel Distributed Processing Symposium. IEEE, 255--265 . H. Cui, L.Wang, Y J. Xue, Yang, and X. Feng. 2011. Automatic Library Generation for BLAS3 on GPUs. In 2011 IEEE International Parallel Distributed Processing Symposium. IEEE, 255--265."},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the 9th Annual IEEE\/ACM International Symposium on Code Generation and Optimization (CGO '11)","author":"Cui H.","unstructured":"H. Cui , J. Xue , L. Wang , Y. Yang , X. Feng , and D. Fan . 2011. Extendable Patternoriented Optimization Directives . In Proceedings of the 9th Annual IEEE\/ACM International Symposium on Code Generation and Optimization (CGO '11) . IEEE, Chamonix, France, 107--118. H. Cui, J. Xue, L. Wang, Y. Yang, X. Feng, and D. Fan. 2011. Extendable Patternoriented Optimization Directives. In Proceedings of the 9th Annual IEEE\/ACM International Symposium on Code Generation and Optimization (CGO '11). IEEE, Chamonix, France, 107--118."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2355585.2355587"},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the 24th International Symposium on High-Performance Computer Architecture (HPCA '18)","author":"Fan D.","unstructured":"D. Fan , X. Ye , W. Li , and D. Wang . 2018. An Efficient Many-Core Processor for High-Throughput Applications in Datacenters . In Proceedings of the 24th International Symposium on High-Performance Computer Architecture (HPCA '18) . D. Fan, X. Ye, W. Li, and D. Wang. 2018. An Efficient Many-Core Processor for High-Throughput Applications in Datacenters. In Proceedings of the 24th International Symposium on High-Performance Computer Architecture (HPCA '18)."},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of the 31th IEEE International Parallel and Distributed Processing Symposium (IPDPS '17)","author":"Fang J.","unstructured":"J. Fang , H. Fu , W. Zhao , B. Chen , W. Zheng , and G. Yang . 2017. swDNN: A Library for Accelerating Deep Learning Applications on Sunway TaihuLight . In Proceedings of the 31th IEEE International Parallel and Distributed Processing Symposium (IPDPS '17) . IEEE, Florida USA. J. Fang, H. Fu, W. Zhao, B. Chen, W. Zheng, and G. Yang. 2017. swDNN: A Library for Accelerating Deep Learning Applications on Sunway TaihuLight. In Proceedings of the 31th IEEE International Parallel and Distributed Processing Symposium (IPDPS '17). IEEE, Florida USA."},{"key":"e_1_3_2_1_19_1","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis (SC '16)","author":"Fu H.","unstructured":"H. Fu , J. Liao, and et al. 2016. Refactoring and Optimizing the Community Atmosphere Model (CAM) on the Sunway TaihuLight Supercomputer . In Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis (SC '16) . IEEE, Salt Lake City, Utah, USA. H. Fu, J. Liao, and et al. 2016. Refactoring and Optimizing the Community Atmosphere Model (CAM) on the Sunway TaihuLight Supercomputer. In Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis (SC '16). IEEE, Salt Lake City, Utah, USA."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-016-5588-7"},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the 12th Annual IEEE\/ACM International Symposium on Code Generation and Optimization (CGO'14)","author":"Grosser T.","unstructured":"T. Grosser , A. Cohen , J. Holewinski , P. Sadayappan , and S. Verdoolaege . 2014. Hybrid Hexagonal\/Classical Tiling for GPUs . In Proceedings of the 12th Annual IEEE\/ACM International Symposium on Code Generation and Optimization (CGO'14) . ACM, Orlando, FL, USA. T. Grosser, A. Cohen, J. Holewinski, P. Sadayappan, and S. Verdoolaege. 2014. Hybrid Hexagonal\/Classical Tiling for GPUs. In Proceedings of the 12th Annual IEEE\/ACM International Symposium on Code Generation and Optimization (CGO'14). ACM, Orlando, FL, USA."},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the 6thWorkshop on General Purpose Processor Using Graphics Processing Units (GPGPU '13)","author":"Grosser T.","unstructured":"T. Grosser , A. Cohen , P. H. J. Kelly , J. Ramanujam , P. Sadayappan , and S. Verdoolaege . 2013. Split tiling for GPUs: Automatic parallelization using trapezoidal tiles . In Proceedings of the 6thWorkshop on General Purpose Processor Using Graphics Processing Units (GPGPU '13) . ACM, 24--31. T. Grosser, A. Cohen, P. H. J. Kelly, J. Ramanujam, P. Sadayappan, and S. Verdoolaege. 2013. Split tiling for GPUs: Automatic parallelization using trapezoidal tiles. In Proceedings of the 6thWorkshop on General Purpose Processor Using Graphics Processing Units (GPGPU '13). ACM, 24--31."},{"key":"e_1_3_2_1_23_1","unstructured":"Khronos Group. 2018. OpenCL Overview. https:\/\/www.khronos.org\/opencl\/  Khronos Group. 2018. OpenCL Overview. https:\/\/www.khronos.org\/opencl\/"},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of the 23rd International Conference on Supercomputing (ICS '09)","author":"Hartono A.","unstructured":"A. Hartono , M. M. Baskaran , C. Bastoul , A. Cohen , S. Krishnamoorthy , B. Norris , J. Ramanujam , and P. Sadayappan . 2009. Parametric Multi-level Tiling of Imperfectly Nested Loops . In Proceedings of the 23rd International Conference on Supercomputing (ICS '09) . ACM, New York, NY, USA, 147--157. A. Hartono, M. M. Baskaran, C. Bastoul, A. Cohen, S. Krishnamoorthy, B. Norris, J. Ramanujam, and P. Sadayappan. 2009. Parametric Multi-level Tiling of Imperfectly Nested Loops. In Proceedings of the 23rd International Conference on Supercomputing (ICS '09). ACM, New York, NY, USA, 147--157."},{"key":"e_1_3_2_1_25_1","volume-title":"2010 IEEE International Symposium on Parallel Distributed Processing.","author":"Hartono A.","unstructured":"A. Hartono , M. M. Baskaran , J. Ramanujam , and P. Sadayappan . 2010. DynTile: Parametric tiled loop generation for parallel execution on multicore processorss . In 2010 IEEE International Symposium on Parallel Distributed Processing. A. Hartono, M. M. Baskaran, J. Ramanujam, and P. Sadayappan. 2010. DynTile: Parametric tiled loop generation for parallel execution on multicore processorss. In 2010 IEEE International Symposium on Parallel Distributed Processing."},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of the 26th ACM International Conference on Supercomputing (ICS '12)","author":"Holewinski J.","unstructured":"J. Holewinski , L. Pouchet , and P. Sadayappan . 2012. High-performance code generation for stencil computations on GPU architectures . In Proceedings of the 26th ACM International Conference on Supercomputing (ICS '12) . ACM, Taiwan, China, 311--320. J. Holewinski, L. Pouchet, and P. Sadayappan. 2012. High-performance code generation for stencil computations on GPU architectures. In Proceedings of the 26th ACM International Conference on Supercomputing (ICS '12). ACM, Taiwan, China, 311--320."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10766-014-0320-y"},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of 5th InternationalWorkshop on Performance Modeling, Benchmarking and Simulation of High Performance Computer Systems (PMBS'14)","author":"Juckeland G.","unstructured":"G. Juckeland , W. C. Brantley , S. Chandrasekaran, and et al. 2014. SPEC ACCEL: A Standard Application Suite for Measuring Hardware Accelerator Performance . In Proceedings of 5th InternationalWorkshop on Performance Modeling, Benchmarking and Simulation of High Performance Computer Systems (PMBS'14) . Springer, New Orleans, LA, USA, 46--67. G. Juckeland,W. C. Brantley, S. Chandrasekaran, and et al. 2014. SPEC ACCEL: A Standard Application Suite for Measuring Hardware Accelerator Performance. In Proceedings of 5th InternationalWorkshop on Performance Modeling, Benchmarking and Simulation of High Performance Computer Systems (PMBS'14). Springer, New Orleans, LA, USA, 46--67."},{"key":"e_1_3_2_1_29_1","volume-title":"Proceedings of the 18th InternationalWorkshop on High-Level Parallel Programming Models and Supportive Environments (HIPS '13)","author":"Krieger C. D.","unstructured":"C. D. Krieger , M. M. Strout , C. Olschanowsky , A. Stone , S. Guzik , X. Gao , C. Bertolli , P. Kelly , G. Mudalige , B. Van Straalen , and S. Williams . 2013. Loop chaining: A programming abstraction for balancing locality and parallelism . In Proceedings of the 18th InternationalWorkshop on High-Level Parallel Programming Models and Supportive Environments (HIPS '13) . Boston, Massachusetts, USA. C. D. Krieger, M. M. Strout, C. Olschanowsky, A. Stone, S. Guzik, X. Gao, C. Bertolli, P. Kelly, G. Mudalige, B. Van Straalen, and S. Williams. 2013. Loop chaining: A programming abstraction for balancing locality and parallelism. In Proceedings of the 18th InternationalWorkshop on High-Level Parallel Programming Models and Supportive Environments (HIPS '13). Boston, Massachusetts, USA."},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the ACM SIGPLAN 1991 Conference on Programming Language Design and Implementation (PLDI '91)","author":"Lam M. S.","unstructured":"M. S. Lam and M. Wolf . 1991. A Data Locality Optimizing Algorithm . In Proceedings of the ACM SIGPLAN 1991 Conference on Programming Language Design and Implementation (PLDI '91) . ACM, New York, NY, USA, 30--44. M. S. Lam and M. Wolf. 1991. A Data Locality Optimizing Algorithm. In Proceedings of the ACM SIGPLAN 1991 Conference on Programming Language Design and Implementation (PLDI '91). ACM, New York, NY, USA, 30--44."},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the 29th International Conference on Compiler Construction (CC '20)","author":"Lazcano R.","unstructured":"R. Lazcano , D. Madro\u00f1al , E. Juarez , and P. Clauss . 2020. Runtime Multi-versioning and Specialization inside a Memoized Speculative Loop Optimizer . In Proceedings of the 29th International Conference on Compiler Construction (CC '20) . ACM, San Diego, CA, USA. R. Lazcano, D. Madro\u00f1al, E. Juarez, and P. Clauss. 2020. Runtime Multi-versioning and Specialization inside a Memoized Speculative Loop Optimizer. In Proceedings of the 29th International Conference on Compiler Construction (CC '20). ACM, San Diego, CA, USA."},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the 19th International Conference on Parallel Architectures and Compilation Techniques (PACT '10)","author":"Lee J.","unstructured":"J. Lee , J. Kim , S. Seo , S. Kim, and et al. 2010. An OpenCL Framework for Heterogeneous Multicores with Local Memory . In Proceedings of the 19th International Conference on Parallel Architectures and Compilation Techniques (PACT '10) . Vienna, Austria, 193--204. J. Lee, J. Kim, S. Seo, S. Kim, and et al. 2010. An OpenCL Framework for Heterogeneous Multicores with Local Memory. In Proceedings of the 19th International Conference on Parallel Architectures and Compilation Techniques (PACT '10). Vienna, Austria, 193--204."},{"key":"e_1_3_2_1_33_1","volume-title":"Proceedings of the 14th International Conference on Parallel Architectures and Compilation Techniques (PACT '05)","author":"Li L.","unstructured":"L. Li , L. Gao , and J. Xue . 2005. Memory Coloring: A Compiler Approach for Scratchpad Memory Management . In Proceedings of the 14th International Conference on Parallel Architectures and Compilation Techniques (PACT '05) . L. Li, L. Gao, and J. Xue. 2005. Memory Coloring: A Compiler Approach for Scratchpad Memory Management. In Proceedings of the 14th International Conference on Parallel Architectures and Compilation Techniques (PACT '05)."},{"key":"e_1_3_2_1_34_1","volume-title":"Proceedings of the 12th Asia-Pacific Conference on Advances in Computer Systems Architecture (ACSAC '07)","author":"Li L.","unstructured":"L. Li , H. Wu , H. Feng , and J. Xue . 2007. Towards Data Tiling for Whole Programs in Scratchpad Memory Allocation . In Proceedings of the 12th Asia-Pacific Conference on Advances in Computer Systems Architecture (ACSAC '07) . Miami Beach, Florida, USA. L. Li, H.Wu, H. Feng, and J. Xue. 2007. Towards Data Tiling for Whole Programs in Scratchpad Memory Allocation. In Proceedings of the 12th Asia-Pacific Conference on Advances in Computer Systems Architecture (ACSAC '07). Miami Beach, Florida, USA."},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the 2011 IEEE International Parallel and Distributed Processing Symposium (IPDPS '11)","author":"Lin H.","unstructured":"H. Lin , T. Liu , L. Renganarayana , H. Li , T. Chen , J. K. O'Brilen , and L. Shao . 2011. Automatic Loop Tiling for Direct Memory Access . In Proceedings of the 2011 IEEE International Parallel and Distributed Processing Symposium (IPDPS '11) . IEEE, New Orleans, Louisiana, USA. H. Lin, T. Liu, L. Renganarayana, H. Li, T. Chen, J. K. O'Brilen, and L. Shao. 2011. Automatic Loop Tiling for Direct Memory Access. In Proceedings of the 2011 IEEE International Parallel and Distributed Processing Symposium (IPDPS '11). IEEE, New Orleans, Louisiana, USA."},{"key":"e_1_3_2_1_36_1","volume-title":"Proceedings of the 31th IEEE International Parallel and Distributed Processing Symposium (IPDPS '17)","author":"Lin H.","unstructured":"H. Lin , X. Tang , B. Yu , Y. Zhuo , W. Chen , J. Zhai , W. Yin , and W. Zheng . 2017. Scalable Graph Traversal on Sunway TaihuLight with Ten Million Cores . In Proceedings of the 31th IEEE International Parallel and Distributed Processing Symposium (IPDPS '17) . IEEE, Florida USA. H. Lin, X. Tang, B. Yu, Y. Zhuo, W. Chen, J. Zhai, W. Yin, and W. Zheng. 2017. Scalable Graph Traversal on Sunway TaihuLight with Ten Million Cores. In Proceedings of the 31th IEEE International Parallel and Distributed Processing Symposium (IPDPS '17). IEEE, Florida USA."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2007.22"},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the 32nd ACM International Conference on Supercomputing (ICS '18)","author":"Liu C.","unstructured":"C. Liu , B. Xie , X. Liu , W. Xue , H. Yang , and X. Liu . 2018. Towards Efficient SpMV on Sunway Many-core Architectures . In Proceedings of the 32nd ACM International Conference on Supercomputing (ICS '18) . ACM, Beijing, China. C. Liu, B. Xie, X. Liu, W. Xue, H. Yang, and X. Liu. 2018. Towards Efficient SpMV on Sunway Many-core Architectures. In Proceedings of the 32nd ACM International Conference on Supercomputing (ICS '18). ACM, Beijing, China."},{"key":"e_1_3_2_1_39_1","volume-title":"Proceedings of the 9th Annual IEEE\/ACM International Symposium on Code Generation and Optimization (CGO '11)","author":"Liu J.","unstructured":"J. Liu , Y. Zhang , W. Ding , and M. T. Kandemir . 2011. On-chip cache hierarchyaware tile scheduling form ulticore machines . In Proceedings of the 9th Annual IEEE\/ACM International Symposium on Code Generation and Optimization (CGO '11) . ACM, Chamonix, France, 161--170. J. Liu, Y. Zhang, W. Ding, and M. T. Kandemir. 2011. On-chip cache hierarchyaware tile scheduling form ulticore machines. In Proceedings of the 9th Annual IEEE\/ACM International Symposium on Code Generation and Optimization (CGO '11). ACM, Chamonix, France, 161--170."},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of the 23rd international conference on Supercomputing (ISC '09)","author":"Liu T.","unstructured":"T. Liu , H. Lin , T. Chen , J. K. O'Brilen , and L. Shao . 2009. DBDB: optimizing DMATransfer for the cell be architecture . In Proceedings of the 23rd international conference on Supercomputing (ISC '09) . ACM, New York, NY, USA, 36--45. T. Liu, H. Lin, T. Chen, J. K. O'Brilen, and L. Shao. 2009. DBDB: optimizing DMATransfer for the cell be architecture. In Proceedings of the 23rd international conference on Supercomputing (ISC '09). ACM, New York, NY, USA, 36--45."},{"key":"e_1_3_2_1_41_1","volume-title":"Proceedings of the 28th International Conference on Compiler Construction (CC'19)","author":"Liu Y.","unstructured":"Y. Liu , L. Huang , M. Wu , H. Cui , F. Lv , X. Feng , and J. Xue . 2019. PPOpenCL: A Performance-Portable OpenCL Compiler with Host and Kernel Thread Code Fusion . In Proceedings of the 28th International Conference on Compiler Construction (CC'19) . ACM, Washington, DC, USA, 2--16. Y. Liu, L. Huang, M. Wu, H. Cui, F. Lv, X. Feng, and J. Xue. 2019. PPOpenCL: A Performance-Portable OpenCL Compiler with Host and Kernel Thread Code Fusion. In Proceedings of the 28th International Conference on Compiler Construction (CC'19). ACM, Washington, DC, USA, 2--16."},{"key":"e_1_3_2_1_42_1","volume-title":"Optimal Tile Size Selection Problem Using Machine Learning. In 2012 11th International Conference on Machine Learning and Applications","volume":"2","author":"Malik A. M.","year":"2012","unstructured":"A. M. Malik . 2012 . Optimal Tile Size Selection Problem Using Machine Learning. In 2012 11th International Conference on Machine Learning and Applications , Vol. 2 . 275--280. A. M. Malik. 2012. Optimal Tile Size Selection Problem Using Machine Learning. In 2012 11th International Conference on Machine Learning and Applications, Vol. 2. 275--280."},{"key":"e_1_3_2_1_43_1","volume-title":"Proceedings of the 2016 International Conference on Supercomputing (ISC '16)","author":"Mehta S.","unstructured":"S. Mehta , R. Garg , N. Trivedi , and P. Yew . 2016. Leveraging Prefetching to Boost Performance of Tiled Codes . In Proceedings of the 2016 International Conference on Supercomputing (ISC '16) . ACM, New York, NY, USA. S. Mehta, R. Garg, N. Trivedi, and P. Yew. 2016. Leveraging Prefetching to Boost Performance of Tiled Codes. In Proceedings of the 2016 International Conference on Supercomputing (ISC '16). ACM, New York, NY, USA."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3314221.3314646"},{"key":"e_1_3_2_1_45_1","volume-title":"Proceedings of the 21th Conference on High Performance Computing Networking, Storage and Analysis (SC '09)","author":"Mohiyuddin M.","unstructured":"M. Mohiyuddin , M. Hoemmen , J. Demmel , and K. Yelick . 2009. Minimizing communication in sparse matrix solvers . In Proceedings of the 21th Conference on High Performance Computing Networking, Storage and Analysis (SC '09) . IEEE, Portland, Oregon, USA. M. Mohiyuddin, M. Hoemmen, J. Demmel, and K. Yelick. 2009. Minimizing communication in sparse matrix solvers. In Proceedings of the 21th Conference on High Performance Computing Networking, Storage and Analysis (SC '09). IEEE, Portland, Oregon, USA."},{"key":"e_1_3_2_1_46_1","volume-title":"Proceedings of the Twentieth International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS '15)","author":"Mullapudi R. T.","unstructured":"R. T. Mullapudi , V. Vasista , and U. Bondhugula . 2015. Automatic optimization for image processing pipelines . In Proceedings of the Twentieth International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS '15) . ACM, Istanbul, Turkey, 429--443. R. T. Mullapudi, V. Vasista, and U. Bondhugula. 2015. Automatic optimization for image processing pipelines. In Proceedings of the Twentieth International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS '15). ACM, Istanbul, Turkey, 429--443."},{"key":"e_1_3_2_1_47_1","volume-title":"Proceedings of the 1997 European Conference on Design and Test (EDTC '97)","author":"Panda P. R.","unstructured":"P. R. Panda , N. D. Dutt , and A. Nicolau . 1997. Efficient Utilization of Scratch-Pad Memory in Embedded Processor Applications . In Proceedings of the 1997 European Conference on Design and Test (EDTC '97) . IEEE Computer Society, USA, 7. P. R. Panda, N. D. Dutt, and A. Nicolau. 1997. Efficient Utilization of Scratch-Pad Memory in Embedded Processor Applications. In Proceedings of the 1997 European Conference on Design and Test (EDTC '97). IEEE Computer Society, USA, 7."},{"key":"e_1_3_2_1_48_1","volume-title":"Proceedings of the 2005 International Conference on Integrated Circuit Design and Technology (ICICDT '05)","author":"Pham D.","unstructured":"D. Pham , S. Asano , M. Bolliger , M. N. Day , H. P. Hofstee , C. Johns , J. Kahle , A. Kameyama , and J. Keaty . 2005. The design and implementation of a firstgeneration CELL processor - a multi-core SoC . In Proceedings of the 2005 International Conference on Integrated Circuit Design and Technology (ICICDT '05) . IEEE, Austin, TX, USA. D. Pham, S. Asano, M. Bolliger, M. N. Day, H. P. Hofstee, C. Johns, J. Kahle, A. Kameyama, and J. Keaty. 2005. The design and implementation of a firstgeneration CELL processor - a multi-core SoC. In Proceedings of the 2005 International Conference on Integrated Circuit Design and Technology (ICICDT '05). IEEE, Austin, TX, USA."},{"key":"e_1_3_2_1_49_1","volume-title":"5th International Workshop on Automatic Performance Tuning.","author":"Rahman M.","unstructured":"M. Rahman , L. Pouchet , and P. Sadayappan . 2010. Neural networks assisted tile size selection . In 5th International Workshop on Automatic Performance Tuning. M. Rahman, L. Pouchet, and P. Sadayappan. 2010. Neural networks assisted tile size selection. In 5th International Workshop on Automatic Performance Tuning."},{"key":"e_1_3_2_1_50_1","volume-title":"Proceedings of the 8th Workshop on General Purpose Processing Using GPUs (GPGPU '15)","author":"Ravishankar M.","unstructured":"M. Ravishankar , J. Holewinski , and V. G. Forma . 2015. A DSL for image processing applications to target GPUs and multi-core CPUs . In Proceedings of the 8th Workshop on General Purpose Processing Using GPUs (GPGPU '15) . ACM, 109--120. M. Ravishankar, J. Holewinski, and V. G. Forma. 2015. A DSL for image processing applications to target GPUs and multi-core CPUs. In Proceedings of the 8th Workshop on General Purpose Processing Using GPUs (GPGPU '15). ACM, 109--120."},{"key":"e_1_3_2_1_51_1","volume-title":"Proceedings of the 25th International Conference on Parallel Architectures and Compilation Techniques (PACT '16)","author":"Rawat P. S.","unstructured":"P. S. Rawat , C. Hong , M. Ravishankar , V. Grover , L. Pouchet , A. Rountev , and P. Sadayappan . 2016. Resource Conscious Reuse-Driven Tiling for GPUs . In Proceedings of the 25th International Conference on Parallel Architectures and Compilation Techniques (PACT '16) . Haifa, Israel. P. S. Rawat, C. Hong, M. Ravishankar, V. Grover, L. Pouchet, A. Rountev, and P. Sadayappan. 2016. Resource Conscious Reuse-Driven Tiling for GPUs. In Proceedings of the 25th International Conference on Parallel Architectures and Compilation Techniques (PACT '16). Haifa, Israel."},{"key":"e_1_3_2_1_52_1","volume-title":"Proceedings of the 28th ACM SIGPLAN Conference on Programming Language Design and Implementation (PLDI '07)","author":"Renganarayanan L.","unstructured":"L. Renganarayanan , D. Kim , S. Rajopadhye , and M. M. Strout . 2007. Parameterized Tiled Loops for Free . In Proceedings of the 28th ACM SIGPLAN Conference on Programming Language Design and Implementation (PLDI '07) . ACM, New York, NY, USA. L. Renganarayanan, D. Kim, S. Rajopadhye, and M. M. Strout. 2007. Parameterized Tiled Loops for Free. In Proceedings of the 28th ACM SIGPLAN Conference on Programming Language Design and Implementation (PLDI '07). ACM, New York, NY, USA."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3293449"},{"key":"e_1_3_2_1_54_1","volume-title":"Proceedings of the 16th International Symposium on Low Power Electronics and Design (ISLPED '10)","author":"Seo S.","unstructured":"S. Seo , R. G. Dreslinski , M. Woh , C. Chakrabarti , S. Mahlke , and T. Mudge . 2010. Diet SODA: A Power-Efficient Processor for Digital Camerasg . In Proceedings of the 16th International Symposium on Low Power Electronics and Design (ISLPED '10) . ACM, Austin, Texas, USA. S. Seo, R. G. Dreslinski, M. Woh, C. Chakrabarti, S. Mahlke, and T. Mudge. 2010. Diet SODA: A Power-Efficient Processor for Digital Camerasg. In Proceedings of the 16th International Symposium on Low Power Electronics and Design (ISLPED '10). ACM, Austin, Texas, USA."},{"key":"e_1_3_2_1_55_1","volume-title":"HPVM: A Portable Virtual Instruction Set for Heterogeneous Parallel Systems. https:\/\/arxiv.org\/pdf\/1611. 00860.pdf","author":"Srivastava P.","year":"2016","unstructured":"P. Srivastava , M. Kotsifakou , and V. Adve . 2016 . HPVM: A Portable Virtual Instruction Set for Heterogeneous Parallel Systems. https:\/\/arxiv.org\/pdf\/1611. 00860.pdf P. Srivastava, M. Kotsifakou, and V. Adve. 2016. HPVM: A Portable Virtual Instruction Set for Heterogeneous Parallel Systems. https:\/\/arxiv.org\/pdf\/1611. 00860.pdf"},{"key":"e_1_3_2_1_56_1","volume-title":"Proceedings of the 29th ACM SIGPLAN Conference on Programming Language Design and Implementation (PLDI '03)","author":"Strout M. M.","unstructured":"M. M. Strout , L. Carter ,, and J. Ferrante . 2003. Compile-time composition of run-time data and iteration reorderings . In Proceedings of the 29th ACM SIGPLAN Conference on Programming Language Design and Implementation (PLDI '03) . ACM, New York, NY, USA. M. M. Strout, L. Carter,, and J. Ferrante. 2003. Compile-time composition of run-time data and iteration reorderings. In Proceedings of the 29th ACM SIGPLAN Conference on Programming Language Design and Implementation (PLDI '03). ACM, New York, NY, USA."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342004041294"},{"key":"e_1_3_2_1_58_1","volume-title":"Proceedings of the 28th IEEE International Parallel and Distributed Processing Symposium (IPDPS '14)","author":"Strout M. M.","unstructured":"M. M. Strout , F. Luporini , C. D. Krieger , and C. Bertolli . 2014. Generalizing Runtime Tiling with the Loop Chain Abstraction . In Proceedings of the 28th IEEE International Parallel and Distributed Processing Symposium (IPDPS '14) . IEEE, New Orleans, Louisiana USA. M. M. Strout, F. Luporini, C. D. Krieger, and C. Bertolli. 2014. Generalizing Runtime Tiling with the Loop Chain Abstraction. In Proceedings of the 28th IEEE International Parallel and Distributed Processing Symposium (IPDPS '14). IEEE, New Orleans, Louisiana USA."},{"key":"e_1_3_2_1_59_1","volume-title":"Proceedings of the Twenty-third Annual ACM Symposium on Parallelism in Algorithms and Architectures (SPAA '11)","author":"Tang Y.","unstructured":"Y. Tang , R. A. Chowdhury , B. C. Kuszmaul , C. Luk , and C. E. Leiserson . 2011. The pochoir stencil compiler . In Proceedings of the Twenty-third Annual ACM Symposium on Parallelism in Algorithms and Architectures (SPAA '11) . ACM, New York, NY, USA, 117--128. Y. Tang, R. A. Chowdhury, B. C. Kuszmaul, C. Luk, and C. E. Leiserson. 2011. The pochoir stencil compiler. In Proceedings of the Twenty-third Annual ACM Symposium on Parallelism in Algorithms and Architectures (SPAA '11). ACM, New York, NY, USA, 117--128."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/2813885.2738003"},{"key":"e_1_3_2_1_61_1","volume-title":"Proceedings of the 23rd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (PPoPP '18)","author":"Wang X.","unstructured":"X. Wang , W. Liu , W. Xue , and L. Wu . 2018. swSpTRSV: a Fast Sparse Triangular Solve with Sparse Level Tile Layout on Sunway Architectures . In Proceedings of the 23rd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (PPoPP '18) . ACM, V\u00f6sendorf\/Wien, Austria. X. Wang, W. Liu, W. Xue, and L. Wu. 2018. swSpTRSV: a Fast Sparse Triangular Solve with Sparse Level Tile Layout on Sunway Architectures. In Proceedings of the 23rd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (PPoPP '18). ACM, V\u00f6sendorf\/Wien, Austria."},{"key":"e_1_3_2_1_62_1","volume-title":"Proceedings of the 47th International Conference on Parallel Processing (ICPP '18)","author":"Wang X.","unstructured":"X. Wang , P. Xu , W. Xue , Y. Ao , C. Yang , H. Fu , L. Gan , G. Yang , and W. Zheng . 2018. A Fast Sparse Triangular Solver for Structured-grid Problems on Sunway Many-core Processor SW26010 . In Proceedings of the 47th International Conference on Parallel Processing (ICPP '18) . ACM, Eugene, OR, USA. X. Wang, P. Xu, W. Xue, Y. Ao, C. Yang, H. Fu, L. Gan, G. Yang, and W. Zheng. 2018. A Fast Sparse Triangular Solver for Structured-grid Problems on Sunway Many-core Processor SW26010. In Proceedings of the 47th International Conference on Parallel Processing (ICPP '18). ACM, Eugene, OR, USA."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-8191(00)00087-9"},{"key":"e_1_3_2_1_64_1","volume-title":"Proceedings of the 31th IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW '17)","author":"Xu Z.","unstructured":"Z. Xu , J. Lin , and S. Matsuoka . 2017. Benchmarking SW26010 Many-core Processor . In Proceedings of the 31th IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW '17) . IEEE, Florida USA. Z. Xu, J. Lin, and S. Matsuoka. 2017. Benchmarking SW26010 Many-core Processor. In Proceedings of the 31th IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW '17). IEEE, Florida USA."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1006\/jpdc.1997.1310"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1142\/S0129626497000401"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.5555\/353939"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.5555\/608716.608764"},{"key":"e_1_3_2_1_69_1","volume-title":"Proceedings of the 2005 International Conference on Parallel Processing (ICPP'05)","author":"Xue J.","unstructured":"J. Xue , Q. Huang , and M. Guo . 2005. Enabling loop fusion and tiling for cache performance by fixing fusion-preventing data dependences . In Proceedings of the 2005 International Conference on Parallel Processing (ICPP'05) . 107--115. J. Xue, Q. Huang, and M. Guo. 2005. Enabling loop fusion and tiling for cache performance by fixing fusion-preventing data dependences. In Proceedings of the 2005 International Conference on Parallel Processing (ICPP'05). 107--115."},{"key":"e_1_3_2_1_70_1","volume-title":"Proceedings of the ACM SIGPLAN 2003 Conference on Programming Language Design and Implementation (PLDI '03)","author":"Yotov K.","unstructured":"K. Yotov , X. Li , G. Ren , M. Cibulskis , G. DeJong , M. Garzaran , D. Padua , K. Pingali , P. Stodghill , and P. Wu . 2003. A Comparison of Empirical and Modeldriven Optimization . In Proceedings of the ACM SIGPLAN 2003 Conference on Programming Language Design and Implementation (PLDI '03) . ACM, New York, NY, USA, 63--76. K. Yotov, X. Li, G. Ren, M. Cibulskis, G. DeJong, M. Garzaran, D. Padua, K. Pingali, P. Stodghill, and P. Wu. 2003. A Comparison of Empirical and Modeldriven Optimization. In Proceedings of the ACM SIGPLAN 2003 Conference on Programming Language Design and Implementation (PLDI '03). ACM, New York, NY, USA, 63--76."},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1145\/1772954.1772982"},{"key":"e_1_3_2_1_72_1","volume-title":"MOCL: An Efficient OpenCL Implementation for the Matrix-2000 Architecture. In Computing Frontiers Conference (CF'18)","author":"Zhang P.","unstructured":"P. Zhang , J. Fang , C. Yang , T. Tang , C. Huang , and Z. Wang . 2018 . MOCL: An Efficient OpenCL Implementation for the Matrix-2000 Architecture. In Computing Frontiers Conference (CF'18) . ACM, Ischia, Italy, 10. P. Zhang, J. Fang, C. Yang, T. Tang, C. Huang, and Z. Wang. 2018. MOCL: An Efficient OpenCL Implementation for the Matrix-2000 Architecture. In Computing Frontiers Conference (CF'18). ACM, Ischia, Italy, 10."},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"crossref","DOI":"10.1145\/3369382","article-title":"Flextended Tiles: A Flexible Extension of Overlapped Tiles for Polyhedral Compilation","volume":"16","author":"Zhao J.","year":"2019","unstructured":"J. Zhao and A. Cohen . 2019 . Flextended Tiles: A Flexible Extension of Overlapped Tiles for Polyhedral Compilation . ACM Transactions on Architecture and Code Optimization 16 , 4 (2019). J. Zhao and A. Cohen. 2019. Flextended Tiles: A Flexible Extension of Overlapped Tiles for Polyhedral Compilation. ACM Transactions on Architecture and Code Optimization 16, 4 (2019).","journal-title":"ACM Transactions on Architecture and Code Optimization"},{"key":"e_1_3_2_1_74_1","volume-title":"Proceedings of the 32nd International Conference on Supercomputing (ICS '18)","author":"Zhao J.","unstructured":"J. Zhao , H. Cui , Y. Zhang , J. Xue , and X. Feng . 2018. Revisiting Loop Tiling for Datacenters: Live and Let Live . In Proceedings of the 32nd International Conference on Supercomputing (ICS '18) . ACM, Beijing, China. J. Zhao, H. Cui, Y. Zhang, J. Xue, and X. Feng. 2018. Revisiting Loop Tiling for Datacenters: Live and Let Live. In Proceedings of the 32nd International Conference on Supercomputing (ICS '18). ACM, Beijing, China."},{"key":"e_1_3_2_1_75_1","volume-title":"Proceedings of the 18th International Conference on High Performance Computing and Communications. IEEE","author":"Zhao M.","unstructured":"M. Zhao , R. Liu , Y. Liu , K. Song , and D. Qian . 2016. Parallel Image Processing on the Sunway Many-core Processor . In Proceedings of the 18th International Conference on High Performance Computing and Communications. IEEE , Sydney, Australia. M. Zhao, R. Liu, Y. Liu, K. Song, and D. Qian. 2016. Parallel Image Processing on the Sunway Many-core Processor. In Proceedings of the 18th International Conference on High Performance Computing and Communications. IEEE, Sydney, Australia."},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.1145\/3177885"}],"event":{"name":"PACT '20: International Conference on Parallel Architectures and Compilation Techniques","location":"Virtual Event GA USA","acronym":"PACT '20","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the ACM International Conference on Parallel Architectures and Compilation Techniques"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3410463.3414637","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3410463.3414637","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T21:31:51Z","timestamp":1750195911000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3410463.3414637"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,9,30]]},"references-count":75,"alternative-id":["10.1145\/3410463.3414637","10.1145\/3410463"],"URL":"https:\/\/doi.org\/10.1145\/3410463.3414637","relation":{},"subject":[],"published":{"date-parts":[[2020,9,30]]},"assertion":[{"value":"2020-09-30","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}