{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,9]],"date-time":"2026-05-09T11:08:36Z","timestamp":1778324916042,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,2,24]],"date-time":"2018-02-24T00:00:00Z","timestamp":1519430400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["1464157"],"award-info":[{"award-number":["1464157"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,2,24]]},"DOI":"10.1145\/3178442.3178445","type":"proceedings-article","created":{"date-parts":[[2018,2,16]],"date-time":"2018-02-16T16:01:58Z","timestamp":1518796918000},"page":"21-30","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["An Evaluation of Vectorization and Cache Reuse Tradeoffs on Modern CPUs"],"prefix":"10.1145","author":[{"given":"Du","family":"Shen","sequence":"first","affiliation":[{"name":"Department of Computer Science, College of William and Mary, Williamsburg, Virgnia, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Milind","family":"Chabbi","sequence":"additional","affiliation":[{"name":"Baidu Research and Scalable Machines Research, California, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xu","family":"Liu","sequence":"additional","affiliation":[{"name":"Department of Computer Science, College of William and Mary, Williamsburg, Virgnia, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2018,2,24]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"SPEC CPU2006","year":"2006","unstructured":"2006. SPEC CPU2006 . http:\/\/www.spec.org\/cpu 2006 \/. (2006). 2006. SPEC CPU2006. http:\/\/www.spec.org\/cpu2006\/. (2006)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/997163.997196"},{"key":"e_1_3_2_1_3_1","volume-title":"Technology Update: The Scalable Vector Extension (SVE) for the ARMv8-A architecture.","author":"ARM Corp.","year":"2016","unstructured":"ARM Corp. 2016 . Technology Update: The Scalable Vector Extension (SVE) for the ARMv8-A architecture. (2016). ARM Corp. 2016. Technology Update: The Scalable Vector Extension (SVE) for the ARMv8-A architecture. (2016)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/774789.774805"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/1071690.1064232"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/MC.2009.57"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2004.21"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1016\/0743-7315(88)90002-0"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2007.32"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/2000417.2000421"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/207110.207162"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/314403.314414"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/2568058.2568069"},{"key":"e_1_3_2_1_15_1","volume-title":"Proc. of the 2014 Intl. Conf. on Timely Results in Operating Systems (TRIOS'14)","author":"Farooqui Naila","year":"2014","unstructured":"Naila Farooqui , Christopher J. Rossbach , Yuan Yu , and Karsten Schwan . 2014 . Leo: A Pro file-driven Dynamic Optimization Framework for GPU Applications . In Proc. of the 2014 Intl. Conf. on Timely Results in Operating Systems (TRIOS'14) . USENIX Association, Berkeley, CA, USA, 5--5. http:\/\/dl.acm.org\/citation.cfm?id=2750315.2750320 Naila Farooqui, Christopher J. Rossbach, Yuan Yu, and Karsten Schwan. 2014. Leo: A Pro file-driven Dynamic Optimization Framework for GPU Applications. In Proc. of the 2014 Intl. Conf. on Timely Results in Operating Systems (TRIOS'14). USENIX Association, Berkeley, CA, USA, 5--5. http:\/\/dl.acm.org\/citation.cfm?id=2750315.2750320"},{"key":"e_1_3_2_1_16_1","volume-title":"Exploring and Evaluating Array Layout Restructuration for SIMDization. In The 27th Intl. Workshop on Languages and Compilers for Parallel Computing (LCPC","author":"Christopher","year":"2014","unstructured":"Christopher Haine et al. 2014 . Exploring and Evaluating Array Layout Restructuration for SIMDization. In The 27th Intl. Workshop on Languages and Compilers for Parallel Computing (LCPC 2014 ). Intel Corporation, Hillsboro, United States. https:\/\/hal.inria.fr\/hal-01070467 Christopher Haine et al. 2014. Exploring and Evaluating Array Layout Restructuration for SIMDization. In The 27th Intl. Workshop on Languages and Compilers for Parallel Computing (LCPC 2014). Intel Corporation, Hillsboro, United States. https:\/\/hal.inria.fr\/hal-01070467"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/2345156.2254108"},{"key":"e_1_3_2_1_18_1","volume-title":"Hot Chips: A Symp. on High Performance Chips. http:\/\/www.hotchips.org","author":"IBM Corp.","year":"2013","unstructured":"IBM Corp. 2013 . POWER8 Processor . In Hot Chips: A Symp. on High Performance Chips. http:\/\/www.hotchips.org IBM Corp. 2013. POWER8 Processor. In Hot Chips: A Symp. on High Performance Chips. http:\/\/www.hotchips.org"},{"key":"e_1_3_2_1_19_1","unstructured":"Intel Corp. 2013. Intel Xeon Phi Core Micro-architecture. https:\/\/software.intel.com\/sites\/default\/files\/article\/393195\/intel-xeon-phi-core-micro-architecture.pdf. (2013).  Intel Corp. 2013. Intel Xeon Phi Core Micro-architecture. https:\/\/software.intel.com\/sites\/default\/files\/article\/393195\/intel-xeon-phi-core-micro-architecture.pdf. (2013)."},{"key":"e_1_3_2_1_20_1","unstructured":"Intel Developer Zone. 2017. Gather and Scatter Intrinsics. https:\/\/software.intel.com\/en-us\/node\/513569. (2017).  Intel Developer Zone. 2017. Gather and Scatter Intrinsics. https:\/\/software.intel.com\/en-us\/node\/513569. (2017)."},{"key":"e_1_3_2_1_21_1","volume-title":"Allen","author":"Kennedy Ken","year":"2002","unstructured":"Ken Kennedy and John R . Allen . 2002 . Optimizing Compilers for Modern Architectures: A Dependence-based Approach. Morgan Kaufmann Publishers Inc ., San Francisco, CA, USA. Ken Kennedy and John R. Allen. 2002. Optimizing Compilers for Modern Architectures: A Dependence-based Approach. Morgan Kaufmann Publishers Inc., San Francisco, CA, USA."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2005.35"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-29740-3_17"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/780732.780735"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2006.15"},{"key":"e_1_3_2_1_26_1","unstructured":"Travis Lanier. 2013. Exploring the Design of the Cortex-A15 Processor. (2013).  Travis Lanier. 2013. Exploring the Design of the Cortex-A15 Processor. (2013)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2013.6557169"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/2503210.2503297"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/2628071.2628102"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2011.68"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/2464996.2465014"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/149439.133079"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/258492.258520"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2014.70"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/237090.237140"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/109025.109108"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/2491956.2462176"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/2370816.2370838"},{"key":"e_1_3_2_1_39_1","volume-title":"Languages and Compilers for Parallel Computing.","author":"Rane Ashay","unstructured":"Ashay Rane , Rakesh Krishnaiyer , Chris J. Newburn , James Browne , Leonardo Fialho , and Zakhar Matveev . 2015. Unification of Static and Dynamic Analyses to Enable Vectorization . In Languages and Compilers for Parallel Computing. Vol. 8967 . 367--381. Ashay Rane, Rakesh Krishnaiyer, Chris J. Newburn, James Browne, Leonardo Fialho, and Zakhar Matveev. 2015. Unification of Static and Dynamic Analyses to Enable Vectorization. In Languages and Compilers for Parallel Computing. Vol. 8967. 367--381."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/2737924.2738004"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2013.6494989"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1147\/rd.413.0233"},{"key":"e_1_3_2_1_43_1","volume-title":"Proceedings of the 2000 IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS '00)","author":"Sarkar V.","unstructured":"V. Sarkar and N. Megiddo . 2000. An Analytical Model for Loop Tiling and Its Solution . In Proceedings of the 2000 IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS '00) . IEEE Computer Society, 146--153. http:\/\/dl.acm.org\/citation.cfm?id=1153923.1154542 V. Sarkar and N. Megiddo. 2000. An Analytical Model for Loop Tiling and Its Solution. In Proceedings of the 2000 IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS '00). IEEE Computer Society, 146--153. http:\/\/dl.acm.org\/citation.cfm?id=1153923.1154542"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-28652-0_6"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/285930.286011"},{"key":"e_1_3_2_1_46_1","volume-title":"In Intl. Workshop on Polyhedral Compilation Techniques (IMPACT).","author":"Vasilache Nicolas","year":"2012","unstructured":"Nicolas Vasilache , Benoit Meister , Albert Hartono , Muthu Baskaran , David Wohlford , and Richard Lethin . 2012 . Trading off memory for parallelism quality . In In Intl. Workshop on Polyhedral Compilation Techniques (IMPACT). Nicolas Vasilache, Benoit Meister, Albert Hartono, Muthu Baskaran, David Wohlford, and Richard Lethin. 2012. Trading off memory for parallelism quality. In In Intl. Workshop on Polyhedral Compilation Techniques (IMPACT)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/71.97902"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/1555754.1555761"}],"event":{"name":"PPoPP '18: 23nd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","location":"Vienna Austria","acronym":"PPoPP '18","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages","SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the 9th International Workshop on Programming Models and Applications for Multicores and Manycores"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3178442.3178445","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3178442.3178445","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3178442.3178445","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T01:39:07Z","timestamp":1750210747000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3178442.3178445"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,2,24]]},"references-count":48,"alternative-id":["10.1145\/3178442.3178445","10.1145\/3178442"],"URL":"https:\/\/doi.org\/10.1145\/3178442.3178445","relation":{},"subject":[],"published":{"date-parts":[[2018,2,24]]},"assertion":[{"value":"2018-02-24","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}