{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,28]],"date-time":"2025-05-28T04:18:28Z","timestamp":1748405908247,"version":"3.41.0"},"publisher-location":"Cham","reference-count":27,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319174723"},{"type":"electronic","value":"9783319174730"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2015]]},"DOI":"10.1007\/978-3-319-17473-0_11","type":"book-chapter","created":{"date-parts":[[2015,4,30]],"date-time":"2015-04-30T09:59:39Z","timestamp":1430387979000},"page":"161-175","source":"Crossref","is-referenced-by-count":3,"title":["Jagged Tiling for Intra-tile Parallelism and Fine-Grain Multithreading"],"prefix":"10.1007","author":[{"given":"Sunil","family":"Shrestha","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Joseph","family":"Manzano","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Andres","family":"Marquez","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"John","family":"Feo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guang R.","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,5,1]]},"reference":[{"key":"11_CR1","unstructured":"perf: Linux profiling with performance counters"},{"key":"11_CR2","doi-asserted-by":"crossref","unstructured":"Bandishti, V., Pananilath, I., Bondhugula, U.: Tiling stencil computations to maximize parallelism. In: Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis, SC 2012, Los Alamitos, CA, USA, pp. 40:1\u201340:11 (2012)","DOI":"10.1109\/SC.2012.107"},{"key":"11_CR3","doi-asserted-by":"crossref","unstructured":"Baskaran, M.M., et al.: Automatic data movement and computation mapping for multi-level parallel architectures with explicitly managed memories. In: Proceedings of the 13th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp. 1\u201310. ACM (2008)","DOI":"10.1145\/1345206.1345210"},{"key":"11_CR4","first-page":"10","volume":"2","author":"C Bastoul","year":"2004","unstructured":"Bastoul, C.: Generating loops for scanning polyhedra: cloog users guide. Polyhedron 2, 10 (2004)","journal-title":"Polyhedron"},{"key":"11_CR5","doi-asserted-by":"crossref","unstructured":"Bikshandi, G., et al.: Programming for parallelism and locality with hierarchically tiled arrays. In: Proceedings of the Eleventh ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, PPoPP 2006, pp. 48\u201357. ACM, New York (2006)","DOI":"10.1145\/1122971.1122981"},{"key":"11_CR6","doi-asserted-by":"crossref","unstructured":"Bondhugula, U., Ramanujam, J.: Pluto: a practical and fully automatic polyhedral parallelizer and locality optimizer (2007)","DOI":"10.1145\/1375581.1375595"},{"key":"11_CR7","unstructured":"Intel Open Source Technology Center. Open community runtime (2012)"},{"key":"11_CR8","unstructured":"Cepeda, S.: Optimization and performance tuning for Intel Xeon Phi coprocessors, part 2: understanding and using hardware events (2012)"},{"key":"11_CR9","doi-asserted-by":"crossref","unstructured":"Datta, K., Kamil, S., Williams, S., Oliker, L., Shalf, J., Yelick, K.: Optimization and performance modeling of stencil computations on modern microprocessors. Siam Rev. (2008)","DOI":"10.1137\/070693199"},{"issue":"2","key":"11_CR10","doi-asserted-by":"publisher","first-page":"946","DOI":"10.1007\/s11227-012-0764-z","volume":"62","author":"H Dursun","year":"2012","unstructured":"Dursun, H., et al.: Hierarchical parallelization and optimization of high-order stencil computations on multicore clusters. J. Supercomput. 62(2), 946\u2013966 (2012)","journal-title":"J. Supercomput."},{"issue":"5","key":"11_CR11","doi-asserted-by":"publisher","first-page":"313","DOI":"10.1007\/BF01407835","volume":"21","author":"P Feautrier","year":"1992","unstructured":"Feautrier, P.: Some efficient solutions to the affine scheduling problem. i. one-dimensional time. Int. J. Parallel Program. 21(5), 313\u2013347 (1992)","journal-title":"Int. J. Parallel Program."},{"issue":"6","key":"11_CR12","doi-asserted-by":"publisher","first-page":"389","DOI":"10.1007\/BF01379404","volume":"21","author":"P Feautrier","year":"1992","unstructured":"Feautrier, P.: Some efficient solutions to the affine scheduling problem. part ii. multidimensional time. Int. J. Parallel Program. 21(6), 389\u2013420 (1992)","journal-title":"Int. J. Parallel Program."},{"key":"11_CR13","doi-asserted-by":"crossref","unstructured":"Frigo, M., Leiserson, C.E., Prokop, H., Ramachandran, S.: Cache-oblivious algorithms. In: Proceedings of the 40th Annual Symposium on Foundations of Computer Science, FOCS 1999, p. 285. IEEE Computer Society, Washington, DC (1999)","DOI":"10.1109\/SFFCS.1999.814600"},{"key":"11_CR14","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"839","DOI":"10.1007\/978-3-642-03869-3_78","volume-title":"Euro-Par 2009 Parallel Processing","author":"G Gan","year":"2009","unstructured":"Gan, G., Wang, X., Manzano, J., Gao, G.R.: Tile percolation: an OpenMP tile aware parallelization technique for the cyclops-64 multicore processor. In: Sips, H., Epema, D., Lin, H.-X. (eds.) Euro-Par 2009. LNCS, vol. 5704, pp. 839\u2013850. Springer, Heidelberg (2009)"},{"key":"11_CR15","doi-asserted-by":"crossref","unstructured":"Griebl, M., Lengauer, C., Wetzel, S.: Code generation in the polytope model. In: Proceedings 1998 International Conference on Parallel Architectures and Compilation Techniques, pp. 106\u2013111. IEEE (1998)","DOI":"10.1109\/PACT.1998.727179"},{"key":"11_CR16","doi-asserted-by":"crossref","unstructured":"Grosser, T., Verdoolaege, S., Cohen, A., Sadayappan, P.: The relation between diamond tiling and hexagonal tiling. In: HiStencils 2014, p. 65 (2014)","DOI":"10.1142\/S0129626414410023"},{"key":"11_CR17","doi-asserted-by":"crossref","unstructured":"H\u00f6gstedt, K., Carter, L., Ferrante, J.: Selecting tile shape for minimal execution time. In: Proceedings of the Eleventh Annual ACM Symposium on Parallel Algorithms and Architectures, pp. 201\u2013211. ACM (1999)","DOI":"10.1145\/305619.305641"},{"key":"11_CR18","unstructured":"ET International. Swarm (swift adaptive runtime machine) (2012)"},{"key":"11_CR19","unstructured":"Kim, D., et al.: Physical experimentation with prefetching helper threads on intel\u2019s hyper-threaded processors. In: Proceedings of the International Symposium on Code Generation and Optimization: Feedback-directed and Runtime Optimization, CGO 2004, p. 27. IEEE Computer Society, Washington, DC (2004)"},{"key":"11_CR20","doi-asserted-by":"crossref","unstructured":"Kodukula, I., Ahmed, N., Pingali, K.: Data-centric multi-level blocking, pp. 346\u2013357 (1997)","DOI":"10.1145\/258916.258946"},{"key":"11_CR21","doi-asserted-by":"crossref","unstructured":"Lewis, J., et al.: An automatic prefetching and caching system. In: 2010 IEEE 29th International Performance Computing and Communications Conference (IPCCC), pp. 180\u2013187, December 2010","DOI":"10.1109\/PCCC.2010.5682310"},{"key":"11_CR22","unstructured":"Massachusetts Institute of Technology: Laboratory for Computer Science and D.O.J. Tanguay. Compile-time Loop Splitting for Distributed Memory Multiprocessors. Massachusetts Institute of Technology, Department of Electrical Engineering and Computer Science (1993)"},{"key":"11_CR23","volume-title":"Earth: An Efficient Architecture for Running Threads","author":"KB Theobald","year":"1999","unstructured":"Theobald, K.B.: Earth: An Efficient Architecture for Running Threads. McGill University, Montreal (1999)"},{"key":"11_CR24","unstructured":"Wilde, D.K.: A library for doing polyhedral operations, Technical report (1997)"},{"key":"11_CR25","doi-asserted-by":"crossref","unstructured":"Wolf, M.E., Lam, M.S.: A data locality optimizing algorithm. In: Proceedings of the ACM SIGPLAN 1991 Conference on Programming Language Design and Implementation, PLDI 1991, pp. 30\u201344. ACM, New York (1991)","DOI":"10.1145\/113446.113449"},{"key":"11_CR26","doi-asserted-by":"crossref","unstructured":"Wolfe, M.: More iteration space tiling. In: Proceedings of the 1989 ACM\/IEEE Conference on Supercomputing, Supercomputing 1989, pp. 655\u2013664. ACM, New York (1989)","DOI":"10.1145\/76263.76337"},{"key":"11_CR27","doi-asserted-by":"crossref","unstructured":"Wolfe, M.: Iteration space tiling for memory hierarchies. In: Proceedings of the Third SIAM Conference on Parallel Processing for Scientific Computing, pp. 357\u2013361. Society for Industrial and Applied Mathematics, Philadelphia (1989)","DOI":"10.1145\/76263.76337"}],"container-title":["Lecture Notes in Computer Science","Languages and Compilers for Parallel Computing"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-17473-0_11","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,27]],"date-time":"2025-05-27T18:35:39Z","timestamp":1748370939000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-17473-0_11"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015]]},"ISBN":["9783319174723","9783319174730"],"references-count":27,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-17473-0_11","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2015]]}}}