{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,19]],"date-time":"2025-08-19T09:59:12Z","timestamp":1755597552403,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,8,5]],"date-time":"2019-08-05T00:00:00Z","timestamp":1564963200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,8,5]]},"DOI":"10.1145\/3337821.3337886","type":"proceedings-article","created":{"date-parts":[[2019,7,25]],"date-time":"2019-07-25T12:34:36Z","timestamp":1564058076000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":10,"title":["Compiler-Assisted GPU Thread Throttling for Reduced Cache Contention"],"prefix":"10.1145","author":[{"given":"Hyunjun","family":"Kim","sequence":"first","affiliation":[{"name":"Sungkyunkwan University, Suwon, Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sungin","family":"Hong","sequence":"additional","affiliation":[{"name":"Sungkyunkwan University, Suwon, Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hyeonsu","family":"Lee","sequence":"additional","affiliation":[{"name":"Sungkyunkwan University, Suwon, Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Euiseong","family":"Seo","sequence":"additional","affiliation":[{"name":"Sungkyunkwan University, Suwon, Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hwansoo","family":"Han","sequence":"additional","affiliation":[{"name":"Sungkyunkwan University, Suwon, Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2019,8,5]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Rajeev Alur Joseph Devietti Omar S. Navarro Leija and Nimit Singhania. 2017. GPUDrano: Detecting Uncoalesced Accesses in GPU Programs. In Computer Aided Verification (CAV).  Rajeev Alur Joseph Devietti Omar S. Navarro Leija and Nimit Singhania. 2017. GPUDrano: Detecting Uncoalesced Accesses in GPU Programs. In Computer Aided Verification (CAV).","DOI":"10.1007\/978-3-319-63387-9_25"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Rajeev Alur Joseph Devietti and Nimit Singhania. 2018. Block-Size Independence for GPU Programs. In Static Analysis (SAS).  Rajeev Alur Joseph Devietti and Nimit Singhania. 2018. Block-Size Independence for GPU Programs. In Static Analysis (SAS).","DOI":"10.1007\/978-3-319-99725-4_9"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2015.38"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2014.11"},{"volume-title":"Locality-Aware Software Throttling for Sparse Matrix Operation on GPUs. In 2018 USENIX Annual Technical Conference (ATC).","author":"Chen Yanhao","key":"e_1_3_2_1_6_1","unstructured":"Yanhao Chen , Ari B. Hayes , Chi Zhang , Timothy Salmon , and Eddy Z. Zhang . 2018 . Locality-Aware Software Throttling for Sparse Matrix Operation on GPUs. In 2018 USENIX Annual Technical Conference (ATC). Yanhao Chen, Ari B. Hayes, Chi Zhang, Timothy Salmon, and Eddy Z. Zhang. 2018. Locality-Aware Software Throttling for Sparse Matrix Operation on GPUs. In 2018 USENIX Annual Technical Conference (ATC)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"S. Grauer-Gray L. Xu R. Searles S. Ayalasomayajula and J. Cavazos. 2012. Auto-Tuning a High-Level Language Targeted to GPU Codes. In Innovative Parallel Computing (InPar).  S. Grauer-Gray L. Xu R. Searles S. Ayalasomayajula and J. Cavazos. 2012. Auto-Tuning a High-Level Language Targeted to GPU Codes. In Innovative Parallel Computing (InPar).","DOI":"10.1109\/InPar.2012.6339595"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/2597652.2597685"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2010.107"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/2304576.2304582"},{"volume-title":"MRPB: Memory Request Prioritization for Massively Parallel Processors. In IEEE 20th International Symposium on High Performance Computer Architecture (HPCA).","author":"Jia W.","key":"e_1_3_2_1_11_1","unstructured":"W. Jia , K.A. Shaw , and M. Martonosi . 2014 . MRPB: Memory Request Prioritization for Massively Parallel Processors. In IEEE 20th International Symposium on High Performance Computer Architecture (HPCA). W.Jia, K.A. Shaw, and M. Martonosi. 2014. MRPB: Memory Request Prioritization for Massively Parallel Processors. In IEEE 20th International Symposium on High Performance Computer Architecture (HPCA)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/2451116.2451158"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the 22Nd International Conference on Parallel Architectures and Compilation Techniques (PACT).","author":"Kayiran Onur","year":"2013","unstructured":"Onur Kayiran , Adwait Jog , Mahmut Taylan Kandemir , and Chita Ranjan Das . 2013 . Neither More nor Less: Optimizing Thread-level Parallelism for GPGPUs . In Proceedings of the 22Nd International Conference on Parallel Architectures and Compilation Techniques (PACT). Onur Kayiran, Adwait Jog, Mahmut Taylan Kandemir, and Chita Ranjan Das. 2013. Neither More nor Less: Optimizing Thread-level Parallelism for GPGPUs. In Proceedings of the 22Nd International Conference on Parallel Architectures and Compilation Techniques (PACT)."},{"key":"e_1_3_2_1_14_1","unstructured":"H. Kim S. Hong and H. Han. 2017. Compiler-Assisted Preloading in the Shared Memory for Thread-Dense Memory Requests. Lecture Notes in Computer Science 11027 0 (2017).  H. Kim S. Hong and H. Han. 2017. Compiler-Assisted Preloading in the Shared Memory for Thread-Dense Memory Requests. Lecture Notes in Computer Science 11027 0 (2017)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2024724.2024754"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080239"},{"volume-title":"Improving GPGPU Resource Utilization Through Alternative Thread Block Scheduling. In IEEE 20th International Symposium on High Performance Computer Architecture (HPCA).","author":"Lee M.","key":"e_1_3_2_1_17_1","unstructured":"M. Lee , S. Song , J. Moon , J. Kim , W. Seo , Y. Cho , and S. Ryu . 2014 . Improving GPGPU Resource Utilization Through Alternative Thread Block Scheduling. In IEEE 20th International Symposium on High Performance Computer Architecture (HPCA). M. Lee, S. Song, J. Moon, J. Kim, W. Seo, Y. Cho, and S. Ryu. 2014. Improving GPGPU Resource Utilization Through Alternative Thread Block Scheduling. In IEEE 20th International Symposium on High Performance Computer Architecture (HPCA)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/2628071.2628107"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3037697.3037709"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/2807591.2807606"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/2751205.2751237"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2015.7054184"},{"volume-title":"Priority-Based Cache Allocation in Throughput Processors. In IEEE 21st International Symposium on High Performance Computer Architecture (HPCA).","author":"Li D.","key":"e_1_3_2_1_23_1","unstructured":"D. Li , M. Rhu , D. R. Johnson , M. O'Connor , M. Erez , D. Burger , D. S. Fussell , and S. W. Redder . 2015 . Priority-Based Cache Allocation in Throughput Processors. In IEEE 21st International Symposium on High Performance Computer Architecture (HPCA). D. Li, M. Rhu, D. R. Johnson, M. O'Connor, M. Erez, D. Burger, D. S. Fussell, and S. W. Redder. 2015. Priority-Based Cache Allocation in Throughput Processors. In IEEE 21st International Symposium on High Performance Computer Architecture (HPCA)."},{"key":"e_1_3_2_1_24_1","unstructured":"NVIDIA. 2017. NVIDIA Tesla V100 GPU Architecture: The World's Most Advanced Data Center GPU.  NVIDIA. 2017. NVIDIA Tesla V100 GPU Architecture: The World's Most Advanced Data Center GPU."},{"key":"e_1_3_2_1_25_1","unstructured":"NVIDIA. 2018. CUDA C Best Practice Guide.  NVIDIA. 2018. CUDA C Best Practice Guide."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3061639.3062320"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/1993498.1993548"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2012.16"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/2540708.2540718"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/2682583"},{"volume-title":"IEEE 21st International Symposium on High Performance Computer Architecture (HPCA).","author":"Sethia A.","key":"e_1_3_2_1_31_1","unstructured":"A. Sethia , D. A. Jamshidi , and S. Mahlke . 2015. Mascar: Speeding up GPU Warps by Reducing Memory Pitstops . In IEEE 21st International Symposium on High Performance Computer Architecture (HPCA). A. Sethia, D. A. Jamshidi, and S. Mahlke. 2015. Mascar: Speeding up GPU Warps by Reducing Memory Pitstops. In IEEE 21st International Symposium on High Performance Computer Architecture (HPCA)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.5555\/2755753.2755911"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2751205.2751239"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/2967938.2967947"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.5555\/2561828.2561929"},{"volume-title":"Coordinated Static and Dynamic Cache Bypassing for GPUs. In IEEE 21st International Symposium on High Performance Computer Architecture (HPCA).","author":"Xie X.","key":"e_1_3_2_1_36_1","unstructured":"X. Xie , Y. Liang , Y. Wang , G. Sun , and T. Wang . 2015 . Coordinated Static and Dynamic Cache Bypassing for GPUs. In IEEE 21st International Symposium on High Performance Computer Architecture (HPCA). X. Xie, Y. Liang, Y. Wang, G. Sun, and T. Wang. 2015. Coordinated Static and Dynamic Cache Bypassing for GPUs. In IEEE 21st International Symposium on High Performance Computer Architecture (HPCA)."},{"volume-title":"CIAO: Cache Interference-Aware Throughput-Oriented Architecture and Scheduling for GPUs. In IEEE International Parallel and Distributed Processing Symposium (IPDPS).","author":"Zhang J.","key":"e_1_3_2_1_37_1","unstructured":"J. Zhang , S. Gao , N. S. Kim , and M. Jung . 2018 . CIAO: Cache Interference-Aware Throughput-Oriented Architecture and Scheduling for GPUs. In IEEE International Parallel and Distributed Processing Symposium (IPDPS). J. Zhang, S. Gao, N. S. Kim, and M. Jung. 2018. CIAO: Cache Interference-Aware Throughput-Oriented Architecture and Scheduling for GPUs. In IEEE International Parallel and Distributed Processing Symposium (IPDPS)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3225058.3225104"}],"event":{"name":"ICPP 2019: 48th International Conference on Parallel Processing","sponsor":["University of Tsukuba University of Tsukuba"],"location":"Kyoto Japan","acronym":"ICPP 2019"},"container-title":["Proceedings of the 48th International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3337821.3337886","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3337821.3337886","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T00:25:41Z","timestamp":1750206341000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3337821.3337886"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,8,5]]},"references-count":38,"alternative-id":["10.1145\/3337821.3337886","10.1145\/3337821"],"URL":"https:\/\/doi.org\/10.1145\/3337821.3337886","relation":{},"subject":[],"published":{"date-parts":[[2019,8,5]]},"assertion":[{"value":"2019-08-05","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}