{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T16:46:38Z","timestamp":1742921198981,"version":"3.40.3"},"publisher-location":"Cham","reference-count":24,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319527086"},{"type":"electronic","value":"9783319527093"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-52709-3_12","type":"book-chapter","created":{"date-parts":[[2017,1,23]],"date-time":"2017-01-23T07:13:25Z","timestamp":1485155605000},"page":"137-152","source":"Crossref","is-referenced-by-count":1,"title":["Automatically Optimizing Stencil Computations on Many-Core NUMA Architectures"],"prefix":"10.1007","author":[{"given":"Pei-Hung","family":"Lin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qing","family":"Yi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Daniel","family":"Quinlan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chunhua","family":"Liao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongqing","family":"Yan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,1,24]]},"reference":[{"key":"12_CR1","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"338","DOI":"10.1007\/11945918_35","volume-title":"High Performance Computing - HiPC 2006","author":"J Antony","year":"2006","unstructured":"Antony, J., Janes, P.P., Rendell, A.P.: Exploring thread and memory placement on numa architectures: Solaris and Linux, UltraSPARC\/FirePlane and Opteron\/HyperTransport. In: Robert, Y., Parashar, M., Badrinath, R., Prasanna, V.K. (eds.) HiPC 2006. LNCS, vol. 4297, pp. 338\u2013352. Springer, Heidelberg (2006). doi: 10.1007\/11945918_35"},{"key":"12_CR2","doi-asserted-by":"crossref","unstructured":"Bircsak, J., Craig, P., Crowell, R., Cvetanovic, Z., Harris, J., Nelson, C.A., Offner, C.D.: Extending OpenMP for NUMA machines. In: ACM\/IEEE 2000 Conference Supercomputing, pp. 48\u201348. IEEE (2000)","DOI":"10.1109\/SC.2000.10019"},{"key":"12_CR3","doi-asserted-by":"crossref","unstructured":"Bolosky, W.J., Scott, M.L., Fitzgerald, R.P., Fowler, R.J., Cox, A.L.: NUMA policies and their relation to memory architecture. In: ACM SIGARCH Computer Architecture News, vol. 19, pp. 212\u2013221. ACM (1991)","DOI":"10.1145\/106972.106994"},{"key":"12_CR4","doi-asserted-by":"crossref","unstructured":"Bondhugula, U., Hartono, A., Ramanujan, J., Sadayappan, P.: A practical automatic polyhedral parallelizer and locality optimizer. In: PLDI 2008: Proceedings of the 2008 ACM SIGPLAN Conference on Programming Language Design and Implementation, pp. 101\u2013113, New York, USA (2008)","DOI":"10.1145\/1375581.1375595"},{"key":"12_CR5","unstructured":"Bull, J.M., Johnson, C.: Data distribution, migration and replication on a cc-NUMA architecture. In: Proceedings of the Fourth European workshop on OpenMP (2002)"},{"key":"12_CR6","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"137","DOI":"10.1007\/3-540-44587-0_13","volume-title":"OpenMP Shared Memory Parallel Programming","author":"B Chapman","year":"2001","unstructured":"Chapman, B., Patil, A., Prabhakar, A.: Performance oriented programming for NUMA architechtures. In: Eigenmann, R., Voss, M.J. (eds.) WOMPAT 2001. LNCS, vol. 2104, pp. 137\u2013154. Springer, Heidelberg (2001). doi: 10.1007\/3-540-44587-0_13"},{"key":"12_CR7","doi-asserted-by":"crossref","unstructured":"Christen, M., Schenk, O., Neufeld, E., Messmer, P., Burkhart, H.: Parallel data-locality aware stencil computations on modern micro-architectures. In: IPDPS 2009: Proceedings of the 2009 IEEE International Symposium on Parallel and Distributed Processing, pp. 1\u201310, Washington, DC, USA (2009)","DOI":"10.1109\/IPDPS.2009.5161031"},{"key":"12_CR8","doi-asserted-by":"crossref","unstructured":"Datta, K., Murphy, M., Volkov, V., Williams, S., Carter, J., Oliker, L., Patterson, D., Shalf, J., Yelick, K.: Stencil computation optimization and auto-tuning on state-of-the-art multicore architectures. In: Proceedings of the 2008 ACM\/IEEE Conference on Supercomputing (SC 2008) (2008)","DOI":"10.1109\/SC.2008.5222004"},{"key":"12_CR9","doi-asserted-by":"crossref","unstructured":"Datta, K., Murphy, M., Volkov, V., Williams, S., Carter, J., Oliker, L., Patterson, D., Shalf, J., Yelick, K.: Stencil computation optimization and auto-tuning on state-of-the-art multicore architectures. In: Proceedings of the 2008 ACM\/IEEE conference on Supercomputing, p. 4. IEEE Press (2008)","DOI":"10.1109\/SC.2008.5222004"},{"key":"12_CR10","unstructured":"Datta, K., Williams, S., Volkov, V., Carter, J., Oliker, L., Shalf, J., Yelick, K.: Auto-tuning the 27-point stencil for multicore. In: Proceedings of iWAPT2009: The Fourth International Workshop on Automatic Performance Tuning (2009)"},{"issue":"3","key":"12_CR11","first-page":"169","volume":"18","author":"L Huang","year":"2010","unstructured":"Huang, L., Jin, H., Yi, L., Chapman, B.: Enabling locality-aware computations in OpenMP. Sci. Prog. 18(3), 169\u2013181 (2010)","journal-title":"Sci. Prog."},{"key":"12_CR12","unstructured":"Kleen, A.: A NUMA API for Linux. Novel Inc (2005)"},{"issue":"6","key":"12_CR13","doi-asserted-by":"crossref","first-page":"235","DOI":"10.1145\/1273442.1250761","volume":"42","author":"S Krishnamoorthy","year":"2007","unstructured":"Krishnamoorthy, S., Baskaran, M., Bondhugula, U., Ramanujam, J., Rountev, A., Sadayappan, P.: Effective automatic parallelization of stencil computations. SIGPLAN Not. 42(6), 235\u2013244 (2007)","journal-title":"SIGPLAN Not."},{"key":"12_CR14","doi-asserted-by":"crossref","unstructured":"Li, S., Hoefler, T., Snir, M.: NUMA-aware shared-memory collective communication for MPI. In: Proceedings of the 22nd International Symposium on High-Performance Parallel and Distributed Computing, pp. 85\u201396. ACM (2013)","DOI":"10.1145\/2493123.2462903"},{"key":"12_CR15","doi-asserted-by":"crossref","unstructured":"Liu, L., Li, V.: Improving parallelism and locality with asynchronous algorithms. In: PPoPP 2010: Proceedings of the 15th ACM SIGPLAN symposium on Principles and Practice of Parallel Programming, New York, NY, USA, pp. 213\u2013222. ACM (2010)","DOI":"10.1145\/1693453.1693483"},{"issue":"6","key":"12_CR16","doi-asserted-by":"crossref","first-page":"545","DOI":"10.1109\/TPDS.2003.1206503","volume":"14","author":"A Navarro","year":"2003","unstructured":"Navarro, A., Zapata, E., Padua, D.: Compiler techniques for the distribution of data and computation. IEEE Trans. Parallel Distrib. Syst. 14(6), 545\u2013562 (2003)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"12_CR17","doi-asserted-by":"crossref","unstructured":"Nikolopoulos, D. S., Papatheodorou, T. S., Polychronopoulos, C. D., Labarta, J., et al.: Is data distribution necessary in OpenMP? In: Proceedings of the 2000 ACM\/IEEE Conference on Supercomputing, p. 47. IEEE Computer Society (2000)","DOI":"10.1109\/SC.2000.10025"},{"key":"12_CR18","unstructured":"OpenMP: Simple, portable, scalable SMP programming. http:\/\/www.openmp.org (2006)"},{"key":"12_CR19","unstructured":"Quinlan, D., et al.: ROSE Compiler Infrastructure. http:\/\/www.rosecompiler.org\/"},{"key":"12_CR20","doi-asserted-by":"crossref","unstructured":"Rivera, G., Tseng, C.-W.: Tiling optimizations for 3D scientific computations. In: Supercomputing 2000: Proceedings of the 2000 ACM\/IEEE Conference on Supercomputing, Washington, DC, USA (2000)","DOI":"10.1109\/SC.2000.10015"},{"key":"12_CR21","doi-asserted-by":"crossref","unstructured":"Shaheen, M., Strzodka, R.: NUMA aware iterative stencil computations on many-core systems. In: 2012 IEEE 26th International Parallel and Distributed Processing Symposium (IPDPS), pp. 461\u2013473. IEEE (2012)","DOI":"10.1109\/IPDPS.2012.50"},{"key":"12_CR22","doi-asserted-by":"crossref","unstructured":"Song, Y., Li, Z.: New tiling techniques to improve cache temporal locality. In: PLDI 1999: Proceedings of the ACM SIGPLAN 1999 Conference on Programming Language Design and Implementation, New York, NY, USA, pp. 215\u2013228 (1999)","DOI":"10.1145\/301618.301668"},{"key":"12_CR23","doi-asserted-by":"crossref","unstructured":"Song, Y., Xu, R., Wang, C., Li, Z.: Data locality enhancement by memory reduction. In: Proceedings of the 15th ACM International Conference on Supercomputing, Sorrento, Italy (2001)","DOI":"10.1145\/377792.377806"},{"issue":"6","key":"12_CR24","doi-asserted-by":"crossref","first-page":"675","DOI":"10.1002\/spe.1089","volume":"42","author":"Q Yi","year":"2012","unstructured":"Yi, Q.: POET: a scripting language for applying parameterized source-to-source program transformations. Softw. Pract. Exp. 42(6), 675\u2013706 (2012)","journal-title":"Softw. Pract. Exp."}],"container-title":["Lecture Notes in Computer Science","Languages and Compilers for Parallel Computing"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-52709-3_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,9,17]],"date-time":"2019-09-17T17:59:04Z","timestamp":1568743144000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-52709-3_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319527086","9783319527093"],"references-count":24,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-52709-3_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2017]]}}}