{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T03:48:23Z","timestamp":1782964103559,"version":"3.54.5"},"reference-count":35,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2015,2]]},"DOI":"10.1109\/hpca.2015.7056031","type":"proceedings-article","created":{"date-parts":[[2015,3,10]],"date-time":"2015-03-10T22:13:51Z","timestamp":1426025631000},"page":"174-185","source":"Crossref","is-referenced-by-count":70,"title":["Mascar: Speeding up GPU warps by reducing memory pitstops"],"prefix":"10.1109","author":[{"given":"Ankit","family":"Sethia","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"D. Anoushe","family":"Jamshidi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Scott","family":"Mahlke","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref33","article-title":"Parboil: A revised benchmark suite for scientific and commercial throughput computing","author":"stratton","year":"2012","journal-title":"Technical Report"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2004.71"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2012.6237038"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2005.28"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2014.2359882"},{"key":"ref34","author":"thoziyoor","year":"2008","journal-title":"CACTI 5 1 Technical Report HPL-2008-20"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2010.20"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/1950365.1950409"},{"key":"ref12","first-page":"37","article-title":"Warped-dmr: Light-weight error detection for gpgpu","author":"hyeran","year":"2012","journal-title":"Proceedings of the 45th Annual International Symposium on Microarchitecture"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2014.6835938"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/2485922.2485951"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/2451116.2451158"},{"key":"ref16","first-page":"157","article-title":"Neither more nor less: Optimizing thread-level parallelism for gpgpus","author":"kayiran","year":"2013","journal-title":"Proceedings of the 22nd International Conference on Parallel Architectures and Compilation Techniques"},{"key":"ref17","year":"2010","journal-title":"OpenCL - the Open Standard for Parallel Programming of Heterogeneous Systems"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/L-CA.2011.32"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2012.6168947"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1145\/2540708.2540718"},{"key":"ref4","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/2063384.2063400","article-title":"Cudadma: Optimizing gpu memory bandwidth via warp specialization","author":"bauer","year":"2011","journal-title":"Proceedings of 2011 International Conference for High Performance Computing Networking Storage and Analysis"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2012.16"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2009.4919648"},{"key":"ref6","first-page":"1","author":"che","year":"2010","journal-title":"A characterization of the rodinia benchmark suite with comparison to contemporary cmp workloads"},{"key":"ref29","first-page":"73","article-title":"Optimization principles and application performance evaluation of a multithreaded GPU using CUDA","author":"ryoo","year":"2008","journal-title":"Proc of the 13th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/2000064.2000093"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2007.30"},{"key":"ref2","doi-asserted-by":"crossref","first-page":"416","DOI":"10.1145\/2366231.2337207","article-title":"Staged memory scheduling: Achieving high performance and scalability in heterogeneous systems","author":"ausavarungnirun","year":"2012","journal-title":"Proc Annual International Symposium on Computer Architecture"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2012.18"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2012.6168946"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/2485922.2485964"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/2155620.2155656"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/1815961.1815992"},{"key":"ref24","year":"2011","journal-title":"CUDA C Programming Guide"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2014.6835955"},{"key":"ref26","author":"rogers","year":"0","journal-title":"CCWS Simulator"},{"key":"ref25","first-page":"86","author":"rhu","year":"2013","journal-title":"A locality-aware memory hierarchy for energy-efficient gpu architectures"}],"event":{"name":"2015 IEEE 21st International Symposium on High Performance Computer Architecture (HPCA)","location":"Burlingame, CA, USA","start":{"date-parts":[[2015,2,7]]},"end":{"date-parts":[[2015,2,11]]}},"container-title":["2015 IEEE 21st International Symposium on High Performance Computer Architecture (HPCA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7048058\/7056013\/07056031.pdf?arnumber=7056031","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2017,6,23]],"date-time":"2017-06-23T07:03:59Z","timestamp":1498201439000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7056031\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,2]]},"references-count":35,"URL":"https:\/\/doi.org\/10.1109\/hpca.2015.7056031","relation":{},"subject":[],"published":{"date-parts":[[2015,2]]}}}