{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T21:36:58Z","timestamp":1784065018093,"version":"3.55.0"},"reference-count":50,"publisher":"Association for Computing Machinery (ACM)","issue":"4","license":[{"start":{"date-parts":[[2018,10,10]],"date-time":"2018-10-10T00:00:00Z","timestamp":1539129600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"UK EPSRC","award":["EP\/L00058X\/1, EP\/L016796\/1, EP\/N031768\/1, and EP\/P010040\/1"],"award-info":[{"award-number":["EP\/L00058X\/1, EP\/L016796\/1, EP\/N031768\/1, and EP\/P010040\/1"]}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"crossref","award":["2016M601031"],"award-info":[{"award-number":["2016M601031"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61702297, 91530323, 41661134014, 41504040, and 61361120098"],"award-info":[{"award-number":["61702297, 91530323, 41661134014, 41504040, and 61361120098"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Tsinghua University Initiative Scientific Research Program","award":["20131089356"],"award-info":[{"award-number":["20131089356"]}]},{"name":"National Key Research 8 Development Plan of China","award":["2017YFA0604500 and 2016YFA0602200"],"award-info":[{"award-number":["2017YFA0604500 and 2016YFA0602200"]}]},{"name":"EU Horizon 2020 Research and Innovation Programme","award":["671653"],"award-info":[{"award-number":["671653"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":["ACM Trans. Archit. Code Optim."],"published-print":{"date-parts":[[2018,12,31]]},"abstract":"<jats:p>This article demonstrates an approach for combining general tuning techniques with the POWER8 hardware architecture through optimizing three representative stencil benchmarks. Two typical real-world applications, with kernels similar to those of the winning programs of the Gordon Bell Prize 2016 and 2017, are employed to illustrate algorithm modifications and a combination of hardware-oriented tuning strategies with the application algorithms. This work fills the gap between hardware capability and software performance of the POWER8 processor, and provides useful guidance for optimizing stencil-based scientific applications on POWER systems.<\/jats:p>","DOI":"10.1145\/3264422","type":"journal-article","created":{"date-parts":[[2018,10,10]],"date-time":"2018-10-10T13:30:46Z","timestamp":1539178246000},"page":"1-25","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["Performance Tuning and Analysis for Stencil-Based Applications on POWER8 Processor"],"prefix":"10.1145","volume":"15","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7311-9924","authenticated-orcid":false,"given":"Jingheng","family":"Xu","sequence":"first","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haohuan","family":"Fu","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wen","family":"Shi","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lin","family":"Gan","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuxuan","family":"Li","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wayne","family":"Luk","sequence":"additional","affiliation":[{"name":"Imperial College London, London, UK"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guangwen","family":"Yang","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2018,10,10]]},"reference":[{"key":"e_1_2_1_1_1","volume-title":"International Workshop on Performance Modeling, Benchmarking and Simulation of High Performance Computer Systems. Springer, 24--45","author":"Adinetz Andrew V.","year":"2014"},{"key":"e_1_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2010.144"},{"key":"e_1_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46079-6_13"},{"key":"e_1_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46079-6_20"},{"key":"e_1_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1177\/109434200001400303"},{"key":"e_1_2_1_6_1","volume-title":"Garc\u00eda","author":"Cebri\u00e1n Juan M.","year":"2017"},{"key":"e_1_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.5555\/1413370.1413375"},{"key":"e_1_2_1_8_1","volume-title":"The LINPACK benchmark: Past, present and future. Concurrency and Computation: Practice and experience 15, 9","author":"Dongarra Jack J.","year":"2003"},{"key":"e_1_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/2832087.2832088"},{"key":"e_1_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/2832087.2832088"},{"key":"e_1_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC.2014.6757353"},{"key":"e_1_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICICDT.2014.6838618"},{"key":"e_1_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2013.111"},{"key":"e_1_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126910"},{"key":"e_1_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126909"},{"key":"e_1_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASAP.2016.7760771"},{"key":"e_1_2_1_17_1","volume-title":"Proceedings of the International Conference on Field Programmable Logic and Applications. 1--6.","author":"Gan L."},{"key":"e_1_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2017.3211107"},{"key":"e_1_2_1_19_1","doi-asserted-by":"crossref","unstructured":"L. Gan H. Fu O. Mencer W. Luk and G. Yang. 2016. Chapter four: Data flow computing in geoscience applications. Advances in Computers (2016).  L. Gan H. Fu O. Mencer W. Luk and G. Yang. 2016. Chapter four: Data flow computing in geoscience applications. Advances in Computers (2016).","DOI":"10.1016\/bs.adcom.2016.09.005"},{"key":"e_1_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/PADSW.2014.7097797"},{"key":"e_1_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/FPL.2014.6927462"},{"key":"e_1_2_1_22_1","volume-title":"Robert Enenkel, Pat Haugen, Michael R. Meissner, Alex Mericas, Philipp Oehler","author":"Hall Brian","year":"2014"},{"key":"e_1_2_1_23_1","unstructured":"IBM Wikipedia and Anandtech. References of IBM POWER8. Retrieved from https:\/\/en.wikipedia.org\/wiki\/POWER8 https:\/\/en.wikipedia.org\/wiki\/Power_Architecture https:\/\/www.anandtech.com\/show\/10435\/assessing-ibms-power8-part-1\/2.  IBM Wikipedia and Anandtech. References of IBM POWER8. Retrieved from https:\/\/en.wikipedia.org\/wiki\/POWER8 https:\/\/en.wikipedia.org\/wiki\/Power_Architecture https:\/\/www.anandtech.com\/show\/10435\/assessing-ibms-power8-part-1\/2."},{"key":"e_1_2_1_24_1","volume-title":"Retrieved","author":"Europe Middle East IBM","year":"2014"},{"key":"e_1_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/2807591.2807674"},{"key":"e_1_2_1_26_1","unstructured":"Intel Wikipedia and stuffedcow. References of Intel E5-2697(v2). Retrieved from https:\/\/ark.intel.com\/products\/75283 http:\/\/blog.stuffedcow.net\/2013\/05\/measuring-rob-capacity\/ https:\/\/en.wikipedia.org\/wiki\/Ivy_Bridge_%28microarchitec-ture%29 https:\/\/en.wikipedia.org\/wiki\/List_of_Intel_CPU_microarchitectures https:\/\/en.wikipedia.org\/wiki\/List_of_Intel_Xeon_microprocessors\u201dIvy_Bridge-EP\u201d_(22_nm)_Efficient_Performance_2.  Intel Wikipedia and stuffedcow. References of Intel E5-2697(v2). Retrieved from https:\/\/ark.intel.com\/products\/75283 http:\/\/blog.stuffedcow.net\/2013\/05\/measuring-rob-capacity\/ https:\/\/en.wikipedia.org\/wiki\/Ivy_Bridge_%28microarchitec-ture%29 https:\/\/en.wikipedia.org\/wiki\/List_of_Intel_CPU_microarchitectures https:\/\/en.wikipedia.org\/wiki\/List_of_Intel_Xeon_microprocessors\u201dIvy_Bridge-EP\u201d_(22_nm)_Efficient_Performance_2."},{"key":"e_1_2_1_27_1","unstructured":"Jim Jeffers and James Reinders. 2015. High Performance Parallelism Pearls Volume Two: Multicore and Many-core Programming Approaches. Morgan Kaufmann.   Jim Jeffers and James Reinders. 2015. High Performance Parallelism Pearls Volume Two: Multicore and Many-core Programming Approaches. Morgan Kaufmann."},{"key":"e_1_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/2503210.2503231"},{"key":"e_1_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2010.5470421"},{"key":"e_1_2_1_30_1","unstructured":"Lawrence Livermore National Laboratory. 2017a. Livermore\u2019s next advanced technology high performance computing system. Retrieved from https:\/\/computation.llnl.gov\/computers\/sierra.  Lawrence Livermore National Laboratory. 2017a. Livermore\u2019s next advanced technology high performance computing system. Retrieved from https:\/\/computation.llnl.gov\/computers\/sierra."},{"key":"e_1_2_1_31_1","unstructured":"Oak Ridge National Laboratory. 2017b. Oak Ridge National Laboratory\u2019s next high performance supercomputer. Retrieved from https:\/\/www.olcf.ornl.gov\/summit\/.  Oak Ridge National Laboratory. 2017b. Oak Ridge National Laboratory\u2019s next high performance supercomputer. Retrieved from https:\/\/www.olcf.ornl.gov\/summit\/."},{"key":"e_1_2_1_32_1","volume-title":"Proceedings of the 2016 IEEE International Parallel and Distributed Processing Symposium. IEEE, 263--272","author":"Liu Xing"},{"key":"e_1_2_1_33_1","volume-title":"STREAM: Sustainable memory bandwidth in high performance computers.","author":"McCalpin John D.","year":"1995"},{"key":"e_1_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1147\/JRD.2014.2380197"},{"key":"e_1_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2010.2"},{"key":"e_1_2_1_36_1","volume-title":"SEG Technical Program Expanded Abstracts","author":"Ortigosa Francisco","year":"2008"},{"key":"e_1_2_1_37_1","volume-title":"Proceedings of the 25th Annual International Conference on Computer Science and Software Engineering. IBM Corp., 61--69","author":"Reguly Istv\u00e1n Z."},{"key":"e_1_2_1_38_1","volume-title":"Giles","author":"Reguly Istv\u00e1n Z.","year":"2016"},{"key":"e_1_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.5555\/370049.370403"},{"key":"e_1_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/2807591.2807675"},{"key":"e_1_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1147\/JRD.2014.2376112"},{"key":"e_1_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46079-6_14"},{"key":"e_1_2_1_44_1","volume-title":"Proceedings of the Workshop on Applications for Multi-and Many-Core Processors (A4MMC) at ISCA","author":"Strzodka Robert","year":"2011"},{"key":"e_1_2_1_45_1","volume-title":"Lottery and Stride Scheduling: Flexibile Proportional-share Resource Management","author":"Waldspurger Carl A."},{"key":"e_1_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/1498765.1498785"},{"key":"e_1_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/TrustCom.2016.0217"},{"key":"e_1_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid.2016.30"},{"key":"e_1_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2014.82"},{"key":"e_1_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/2517327.2442518"},{"key":"e_1_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.5555\/3014904.3014912"}],"container-title":["ACM Transactions on Architecture and Code Optimization"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3264422","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3264422","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T08:39:35Z","timestamp":1750235975000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3264422"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,10,10]]},"references-count":50,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2018,12,31]]}},"alternative-id":["10.1145\/3264422"],"URL":"https:\/\/doi.org\/10.1145\/3264422","relation":{},"ISSN":["1544-3566","1544-3973"],"issn-type":[{"value":"1544-3566","type":"print"},{"value":"1544-3973","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,10,10]]},"assertion":[{"value":"2018-02-01","order":0,"name":"received","label":"Received","group":{"name":"publication_history","label":"Publication History"}},{"value":"2018-07-01","order":1,"name":"accepted","label":"Accepted","group":{"name":"publication_history","label":"Publication History"}},{"value":"2018-10-10","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}