{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T04:58:53Z","timestamp":1750309133966,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":18,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,5,7]],"date-time":"2024-05-07T00:00:00Z","timestamp":1715040000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["CNS-1822080"],"award-info":[{"award-number":["CNS-1822080"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,5,7]]},"DOI":"10.1145\/3649153.3649214","type":"proceedings-article","created":{"date-parts":[[2024,7,2]],"date-time":"2024-07-02T10:21:29Z","timestamp":1719915689000},"page":"97-105","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["BLP: Block-Level Pipelining for GPUs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6015-0727","authenticated-orcid":false,"given":"Wu-chun","family":"Feng","sequence":"first","affiliation":[{"name":"Virginia Tech, Blacksburg, Virginia, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0173-5215","authenticated-orcid":false,"given":"Xuewen","family":"Cui","sequence":"additional","affiliation":[{"name":"Virginia Tech, Blacksburg, Virginia, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7234-5743","authenticated-orcid":false,"given":"Thomas","family":"Scogland","sequence":"additional","affiliation":[{"name":"Lawrence Livermore National Lab, Livermore, California, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0339-1006","authenticated-orcid":false,"given":"Bronis","family":"de Supinski","sequence":"additional","affiliation":[{"name":"Lawrence Livermore National Lab, Livermore, California, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,7,2]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"OpenMP ARB \"OpenMP Application Program Interface Version 4.0 \" 2013."},{"key":"e_1_3_2_1_2_1","unstructured":"OpenMP ARB \"OpenMP Application Program Interface Version 5.0 \" 2018."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2013.222"},{"key":"e_1_3_2_1_4_1","first-page":"855","volume-title":"2017 IEEE International. IEEE","author":"Klenk B.","year":"2017","unstructured":"B. Klenk, H. Fr\u00f6ening, H. Eberle, and L. Dennison, \"Relaxations for High-Performance Message Passing on Massively Parallel SIMT Processors,\" in Parallel and Distributed Processing Symposium (IPDPS), 2017 IEEE International. IEEE, 2017, pp. 855--865."},{"key":"e_1_3_2_1_5_1","first-page":"233","volume-title":"ACM","author":"Rossbach C.J.","year":"2011","unstructured":"C.J. Rossbach, J. Currey, M. Silberstein, B. Ray, and E. Witchel, \"PTask: Operating System Abstractions to Manage GPUs as Compute Devices,\" in Proceedings of the Twenty-Third ACM Symposium on Operating Systems Principles. ACM, 2011, pp. 233--248."},{"key":"e_1_3_2_1_6_1","first-page":"407","volume-title":"USA: ACM","author":"Chen G.","year":"2015","unstructured":"G. Chen and X. Shen, \"Free Launch: Optimizing GPU Dynamic Kernel Launches Through Thread Reuse,\" in Proceedings of the 48th International Symposium on Microarchitecture, ser. MICRO-48. New York, NY, USA: ACM, 2015, pp. 407--419. [Online]. Available: http:\/\/doi.acm.org\/10.1145\/2830772.2830818"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2018.022071134"},{"key":"e_1_3_2_1_8_1","first-page":"575","volume-title":"d. Supinski, and W. Feng, \"Directive-based partitioning and pipelining for graphics processing units,\" in 2017 IEEE International Parallel and Distributed Processing Symposium (IPDPS)","author":"Cui X.","year":"2017","unstructured":"X. Cui, T. R. W. Scogland, B. R. d. Supinski, and W. Feng, \"Directive-based partitioning and pipelining for graphics processing units,\" in 2017 IEEE International Parallel and Distributed Processing Symposium (IPDPS), 2017, pp. 575--584."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/358438.349322"},{"key":"e_1_3_2_1_10_1","volume-title":"The Polyhedral Benchmark Suite,\" URL: http:\/\/www.cs. ucla. edu\/pouchet\/software\/polybench","author":"Pouchet L.-N.","year":"2012","unstructured":"L.-N. Pouchet, \"Polybench: The Polyhedral Benchmark Suite,\" URL: http:\/\/www.cs. ucla. edu\/pouchet\/software\/polybench, 2012."},{"key":"e_1_3_2_1_11_1","volume-title":"Digital Image Processing","author":"Gonzalez R. C.","year":"2002","unstructured":"R. C. Gonzalez and R. E. Woods, \"Digital Image Processing,\" 2002."},{"issue":"2","key":"e_1_3_2_1_12_1","first-page":"2596","article-title":"Time Reversal Processing for Source Location in an Urban Environment","volume":"115","author":"Albert D. G.","year":"2005","unstructured":"D. G. Albert, L. Liu, and M. L. Moran, \"Time Reversal Processing for Source Location in an Urban Environment,\" The Journal of the Acoustical Society of America, vol. 115, no. 2, pp. 2596--619, 2005.","journal-title":"The Journal of the Acoustical Society of America"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511596834"},{"key":"e_1_3_2_1_14_1","first-page":"1","article-title":"A Study of Persistent Threads Style GPU Programming for GPGPU workloads,\" in Innovative Parallel Computing-Foundations & Applications of GPU, Manycore, and Heterogeneous Systems (INPAR 2012)","author":"Gupta K.","year":"2012","unstructured":"K. Gupta, J. A. Stuart, and J. D. Owens, \"A Study of Persistent Threads Style GPU Programming for GPGPU workloads,\" in Innovative Parallel Computing-Foundations & Applications of GPU, Manycore, and Heterogeneous Systems (INPAR 2012). IEEE, 2012, pp. 1--14.","journal-title":"IEEE"},{"key":"e_1_3_2_1_15_1","first-page":"144","volume-title":"IEEE 26th International Parallel & Distributed Processing Symposium (IPDPS). IEEE","author":"Scogland T. R.","year":"2012","unstructured":"T. R. Scogland, B. Rountree, W.-c. Feng, and B. R. de Supinski, \"Heterogeneous Task Scheduling for Accelerated OpenMP,\" in IEEE 26th International Parallel & Distributed Processing Symposium (IPDPS). IEEE, 2012, pp. 144--155."},{"volume-title":"Core Task-Size Adapting Runtime,\" IEEE transactions on parallel and distributed systems","author":"Scogland T. R.","key":"e_1_3_2_1_16_1","unstructured":"T. R. Scogland, W.-c. Feng, B. Rountree, and B. R. de Supinski, \"CoreTSAR: Core Task-Size Adapting Runtime,\" IEEE transactions on parallel and distributed systems, vol. 26, no. 11, pp. 2970--2983, 2015."},{"key":"e_1_3_2_1_17_1","volume-title":"ACM","author":"Bueno J.","year":"2013","unstructured":"J. Bueno, X. Martorell, R. M. Badia, E. Ayguad\u00e9, and J. Labarta, \"Implementing OmpSs Support for Regions of Data in Architectures with Multiple Address Spaces,\" in ACM International Conference on Supercomputing. ACM, Jun. 2013."},{"key":"e_1_3_2_1_19_1","first-page":"12","volume-title":"Storage and Analysis. ACM","author":"Bauer M.","year":"2011","unstructured":"M. Bauer, H. Cook, and B. Khailany, \"CudaDMA: Optimizing GPU Memory Bandwidth via Warp Specialization,\" in Proceedings of SC11, the International Conference for High Performance Computing, Networking, Storage and Analysis. ACM, 2011, p. 12."}],"event":{"name":"CF '24: 21st ACM International Conference on Computing Frontiers","sponsor":["SIGMICRO ACM Special Interest Group on Microarchitectural Research and Processing"],"location":"Ischia Italy","acronym":"CF '24"},"container-title":["Proceedings of the 21st ACM International Conference on Computing Frontiers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3649153.3649214","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3649153.3649214","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T22:50:02Z","timestamp":1750287002000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3649153.3649214"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,7]]},"references-count":18,"alternative-id":["10.1145\/3649153.3649214","10.1145\/3649153"],"URL":"https:\/\/doi.org\/10.1145\/3649153.3649214","relation":{},"subject":[],"published":{"date-parts":[[2024,5,7]]},"assertion":[{"value":"2024-07-02","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}