{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T13:51:26Z","timestamp":1760709086650},"publisher-location":"Cham","reference-count":18,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319413204"},{"type":"electronic","value":"9783319413211"}],"license":[{"start":{"date-parts":[[2016,1,1]],"date-time":"2016-01-01T00:00:00Z","timestamp":1451606400000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016]]},"DOI":"10.1007\/978-3-319-41321-1_1","type":"book-chapter","created":{"date-parts":[[2016,6,14]],"date-time":"2016-06-14T10:19:15Z","timestamp":1465899555000},"page":"3-20","source":"Crossref","is-referenced-by-count":1,"title":["An Analytical Model-Based Auto-tuning Framework for Locality-Aware Loop Scheduling"],"prefix":"10.1007","author":[{"given":"Rengan","family":"Xu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sunita","family":"Chandrasekaran","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaonan","family":"Tian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Barbara","family":"Chapman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,6,15]]},"reference":[{"key":"1_CR1","unstructured":"EPCC OpenACC Benchmarks (2015). https:\/\/www.epcc.ed.ac.uk\/research\/computing\/performance-characterisation-and-benchmarking\/epcc-openacc-benchmark-suite"},{"key":"1_CR2","unstructured":"KernelGen Performance Test Suite, December 2015. https:\/\/hpcforge.org\/plugins\/mediawiki\/wiki\/kernelgen\/index.php\/Performance_Test_Suite"},{"key":"1_CR3","unstructured":"OpenACC (2016). http:\/\/www.openacc.org"},{"key":"1_CR4","doi-asserted-by":"crossref","unstructured":"Alm\u00e1si, G., Ca\u015fcaval, C., Padua, D.A.: Calculating stack distances efficiently. In: ACM SIGPLAN Notices, vol. 38, pp. 37\u201343. ACM (2002)","DOI":"10.1145\/773146.773043"},{"key":"1_CR5","unstructured":"Baghsorkhi, S.S., Delahaye, M., Gropp, W.D., Wen-mei, W.H..: Analytical performance prediction for evaluation and tuning of GPGPU applications. In: Workshop on EPHAM2009, in Conjunction with CGO, Citeseer (2009)"},{"key":"1_CR6","unstructured":"Beyls, K., Hollander, E.D.: Reuse distance as a metric for cache behavior. In: Proceedings of the IASTED Conference on Parallel and Distributed Computing and Systems, vol. 14, pp. 350\u2013360 (2001)"},{"issue":"2","key":"1_CR7","doi-asserted-by":"crossref","first-page":"78","DOI":"10.5573\/IEIESPC.2015.4.2.078","volume":"4","author":"KH Choi","year":"2015","unstructured":"Choi, K.H., Kim, S.W.: Study of cache performance on GPGPU. IEIE Trans. Smart Process. Comput. 4(2), 78\u201382 (2015)","journal-title":"IEIE Trans. Smart Process. Comput."},{"key":"1_CR8","doi-asserted-by":"crossref","unstructured":"Cui, X., Chen, Y., Zhang, C., Mei, H.: Auto-tuning dense matrix multiplication for GPGPU with cache. In: IEEE 16th International Conference on Parallel and Distributed Systems (ICPADS), pp. 237\u2013242. IEEE (2010)","DOI":"10.1109\/ICPADS.2010.64"},{"key":"1_CR9","doi-asserted-by":"crossref","unstructured":"Grauer-Gray, S., Xu, L., Searles, R., Ayalasomayajula, S., Cavazos, J.: Auto-tuning a high-level language targeted to GPU codes. In: Innovative Parallel Computing (InPar), pp. 1\u201310. IEEE (2012)","DOI":"10.1109\/InPar.2012.6339595"},{"key":"1_CR10","unstructured":"Hu, Y., Koppelman, D.M., Brandt, S.R., L\u00f6ffler, F.: Model-driven auto-tuning of stencil computations on GPUs. In: Proceedings of the 2nd International Workshop on High-Performance Stencil Computations, pp. 1\u20138 (2015)"},{"key":"1_CR11","doi-asserted-by":"crossref","unstructured":"Lee, H., Brown, K.J., Sujeeth, A.K., Rompf, T., Olukotun, K.: Locality-aware mapping of nested parallel patterns on GPUs. In: 47th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO), pp. 63\u201374. IEEE (2014)","DOI":"10.1109\/MICRO.2014.23"},{"key":"1_CR12","doi-asserted-by":"crossref","unstructured":"Mametjanov, A., Lowell, D., Ma, C.-C., Norris, B.: Autotuning stencil-based computations on GPUs. In: IEEE International Conference on Cluster Computing (CLUSTER), pp. 266\u2013274. IEEE (2012)","DOI":"10.1109\/CLUSTER.2012.46"},{"key":"1_CR13","doi-asserted-by":"crossref","unstructured":"Montgomery, C., Overbey, J.L., Li, X.: Autotuning openACC work distribution via direct search. In: Proceedings of the 2015 XSEDE Conference: Scientific Advancements Enabled by Enhanced Cyberinfrastructure, p. 38. ACM (2015)","DOI":"10.1145\/2792745.2792783"},{"key":"1_CR14","doi-asserted-by":"crossref","unstructured":"Nugteren, C., van den Braak, G.-J., Corporaal, H., Bal, H.: A detailed GPU cache model based on reuse distance theory. In: High Performance Computer Architecture (HPCA), pp. 37\u201348. IEEE (2014)","DOI":"10.1109\/HPCA.2014.6835955"},{"key":"1_CR15","doi-asserted-by":"crossref","unstructured":"Picchi, J., Zhang, W.: Impact of L2 cache locking on GPU performance. In: SoutheastCon, pp. 1\u20134. IEEE (2015)","DOI":"10.1109\/SECON.2015.7133036"},{"key":"1_CR16","series-title":"Lecture Notes in Computer Science","first-page":"224","volume-title":"High Performance Computing for Computational Science-VECPAR","author":"S Siddiqui","year":"2014","unstructured":"Siddiqui, S., AlZayer, F., Feki, S.: Historic learning approach for auto-tuning openACC accelerated scientific applications. VECPAR-2014. LNCS, vol. 8969, pp. 224\u2013235. Springer, Heidelberg (2014)"},{"key":"1_CR17","doi-asserted-by":"crossref","unstructured":"Tang, T., Yang, X., Lin, Y.: Cache miss analysis for GPU programs based on stack distance profile. In: 31st International Conference on Distributed Computing Systems (ICDCS), pp. 623\u2013634. IEEE (2011)","DOI":"10.1109\/ICDCS.2011.16"},{"key":"1_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"crossref","first-page":"105","DOI":"10.1007\/978-3-319-09967-5_6","volume-title":"Languages and Compilers for Parallel Computing","author":"X Tian","year":"2014","unstructured":"Tian, X., Xu, R., Yan, Y., Yun, Z., Chandrasekaran, S., Chapman, B.: Compiling a high-level directive-based programming model for GPGPUs. LCPC 2013. LNCS, vol. 8664, pp. 105\u2013120. Springer International Publishing, New York (2014)"}],"container-title":["Lecture Notes in Computer Science","High Performance Computing"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-41321-1_1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2017,6,24]],"date-time":"2017-06-24T16:20:36Z","timestamp":1498321236000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-41321-1_1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016]]},"ISBN":["9783319413204","9783319413211"],"references-count":18,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-41321-1_1","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2016]]}}}