{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T15:47:43Z","timestamp":1772725663618,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":61,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,2,24]],"date-time":"2018-02-24T00:00:00Z","timestamp":1519430400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,2,24]]},"DOI":"10.1145\/3180270.3180271","type":"proceedings-article","created":{"date-parts":[[2018,2,12]],"date-time":"2018-02-12T14:03:02Z","timestamp":1518444182000},"page":"50-60","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":23,"title":["Oversubscribed Command Queues in GPUs"],"prefix":"10.1145","author":[{"given":"Sooraj","family":"Puthoor","sequence":"first","affiliation":[{"name":"AMD Research"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xulong","family":"Tang","sequence":"additional","affiliation":[{"name":"Penn State"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Joseph","family":"Gross","sequence":"additional","affiliation":[{"name":"AMD Research"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bradford M.","family":"Beckmann","sequence":"additional","affiliation":[{"name":"AMD Research"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2018,2,24]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"AMD. \"Asynchronous shaders\". http:\/\/amd-dev.wpengine.netdna-cdn.com\/wordpress\/media\/2012\/10\/Asynchronous-Shaders-White-Paper-FINAL.pdf  AMD. \"Asynchronous shaders\". http:\/\/amd-dev.wpengine.netdna-cdn.com\/wordpress\/media\/2012\/10\/Asynchronous-Shaders-White-Paper-FINAL.pdf"},{"key":"e_1_3_2_1_2_1","unstructured":"AMD. \"AMD FirePro GPUs\". http:\/\/www.amd.com\/en-us\/innovations\/software-technologies\/apu  AMD. \"AMD FirePro GPUs\". http:\/\/www.amd.com\/en-us\/innovations\/software-technologies\/apu"},{"key":"e_1_3_2_1_3_1","unstructured":"AMD. \"AMD GCN Architecture\". https:\/\/www.amd.com\/Documents\/GCN_Architecture_whitepaper.pdf  AMD. \"AMD GCN Architecture\". https:\/\/www.amd.com\/Documents\/GCN_Architecture_whitepaper.pdf"},{"key":"e_1_3_2_1_4_1","unstructured":"ATMI\n\n  \n  : https:\/\/gpuopen.com\/compute-product\/atmi\/  ATMI: https:\/\/gpuopen.com\/compute-product\/atmi\/"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/IADCC.2009.4808986"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2003.1206502"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"M. Bauer S. Treichler E. Slaughter A. Aiken \"Legion: Expressing Locality and Independence with Logical Regions.\" In the International Conference on Supercomputing 2012   M. Bauer S. Treichler E. Slaughter A. Aiken \"Legion: Expressing Locality and Independence with Logical Regions.\" In the International Conference on Supercomputing 2012","DOI":"10.1109\/SC.2012.71"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/2024716.2024718"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/324133.324234"},{"key":"e_1_3_2_1_10_1","volume-title":"Hot Chips: A Symposium on High Performance Chips (HC26)","author":"Bouvier D."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"N. Brunie S. Collange and G. Diamos \"Simultaneous branch and warp interweaving for sustained GPU performance \" 2012 39th Annual International Symposium on Computer Architecture (ISCA)   N. Brunie S. Collange and G. Diamos \"Simultaneous branch and warp interweaving for sustained GPU performance \" 2012 39th Annual International Symposium on Computer Architecture (ISCA)","DOI":"10.1109\/ISCA.2012.6237005"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/1094811.1094852"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/2830772.2830818"},{"key":"e_1_3_2_1_14_1","unstructured":"N. Christofides \"Graph Theory: An algorithmic Approach.\" 1975.   N. Christofides \"Graph Theory: An algorithmic Approach.\" 1975."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2751205.2751235"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2008.88"},{"key":"e_1_3_2_1_17_1","unstructured":"CUDA streams. https:\/\/devblogs.nvidia.com\/parallelforall\/gpu-pro-tip-cuda-7-streams-simplify-concurrency\/  CUDA streams. https:\/\/devblogs.nvidia.com\/parallelforall\/gpu-pro-tip-cuda-7-streams-simplify-concurrency\/"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPADS.2006.40"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1142\/S0129626411000151"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2010.13"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/277650.277725"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2007.12"},{"key":"e_1_3_2_1_23_1","unstructured":"G. Krishnan D. Bouvier L. Zhang and P. Dongara. \"Energy Efficient Graphics and Multimedia in 28nm Carrizo APU\" In Hot Chips: A Symposium on High Performance Chips (HC27).  G. Krishnan D. Bouvier L. Zhang and P. Dongara. \"Energy Efficient Graphics and Multimedia in 28nm Carrizo APU\" In Hot Chips: A Symposium on High Performance Chips (HC27)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2005.175"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"I. E. Hajj J. Gomez-Luna C. Li L. W. Chang D. Milojicic and W. m. Hwu \"KLAP: Kernel launch aggregation and promotion for optimizing dynamic parallelism \" 2016 49th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO)  I. E. Hajj J. Gomez-Luna C. Li L. W. Chang D. Milojicic and W. m. Hwu \"KLAP: Kernel launch aggregation and promotion for optimizing dynamic parallelism \" 2016 49th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO)","DOI":"10.1109\/MICRO.2016.7783716"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1137\/0218016"},{"key":"e_1_3_2_1_27_1","unstructured":"HSA Foundation. (2016). \"HSA Platform System Architecture Specification\". Version 1.1. http:\/\/www.hsafoundation.com\/standards  HSA Foundation. (2016). \"HSA Platform System Architecture Specification\". Version 1.1. http:\/\/www.hsafoundation.com\/standards"},{"key":"e_1_3_2_1_28_1","unstructured":"HSA Foundation. \"HSA Runtime Programmers Reference Manual. Version 1.1\" (2016). http:\/\/www.hsafoundation.com\/standards  HSA Foundation. \"HSA Runtime Programmers Reference Manual. Version 1.1\" (2016). http:\/\/www.hsafoundation.com\/standards"},{"key":"e_1_3_2_1_29_1","unstructured":"HSA Foundation. (2016). \"HSA Runtime Specification\". Version 1.1. http:\/\/www.hsafoundation.com\/standards  HSA Foundation. (2016). \"HSA Runtime Specification\". Version 1.1. http:\/\/www.hsafoundation.com\/standards"},{"key":"e_1_3_2_1_30_1","unstructured":"ioctl: http:\/\/man7.org\/linux\/man-pages\/man2\/ioctl.2.html  ioctl: http:\/\/man7.org\/linux\/man-pages\/man2\/ioctl.2.html"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/2451116.2451158"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/2676870.2676883"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/1378533.1378575"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/1273440.1250683"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/2628071.2628107"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPADS.2006.37"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/2830772.2830822"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/1105734.1105747"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/2155620.2155656"},{"key":"e_1_3_2_1_40_1","unstructured":"NVIDIA \"DYNAMIC PARALLELISM IN CUDA\" http:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/index.html#cuda-dynamic-parallelism  NVIDIA \"DYNAMIC PARALLELISM IN CUDA\" http:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/index.html#cuda-dynamic-parallelism"},{"key":"e_1_3_2_1_41_1","unstructured":"NVIDIA Tesla GPUs: http:\/\/www.nvidia.com\/object\/tesla-servers.html  NVIDIA Tesla GPUs: http:\/\/www.nvidia.com\/object\/tesla-servers.html"},{"key":"e_1_3_2_1_42_1","unstructured":"NVIDIA \"JP Morgan Speeds Risk Calculations with NVIDIA GPUs \" 2011.  NVIDIA \"JP Morgan Speeds Risk Calculations with NVIDIA GPUs \" 2011."},{"key":"e_1_3_2_1_43_1","unstructured":"OpenMP4.5 Specification. (2015). \"The OpenMP Architecture Review Board\".http:\/\/www.openmp.org\/mp-documents\/openmp-4.5.pdf  OpenMP4.5 Specification. (2015). \"The OpenMP Architecture Review Board\".http:\/\/www.openmp.org\/mp-documents\/openmp-4.5.pdf"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2005.184"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/AIPR.2008.4906458"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","unstructured":"G. Pratx and L. Xing \"GPU Computing in Medical Physics: A Review \" in Medical physics 2011.  G. Pratx and L. Xing \"GPU Computing in Medical Physics: A Review \" in Medical physics 2011.","DOI":"10.1118\/1.3578605"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/2884045.2884052"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/2540708.2540718"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2012.16"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/1736020.1736055"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2008.05.013"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"crossref","unstructured":"X. Tang A. Pattnaik H. Jiang O. Kayiran A. Jog S. Pai M. Ibrahim M. T. Kandemir and C. Das. \"Controlled Kernel Launch for Dynamic Parallelism in GPUs.\" In proceedings of The 23rd International Symposium on High-Performance Computer Architecture (HPCA 2017)  X. Tang A. Pattnaik H. Jiang O. Kayiran A. Jog S. Pai M. Ibrahim M. T. Kandemir and C. Das. \"Controlled Kernel Launch for Dynamic Parallelism in GPUs.\" In proceedings of The 23rd International Symposium on High-Performance Computer Architecture (HPCA 2017)","DOI":"10.1109\/HPCA.2017.14"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10723-012-9237-0"},{"key":"e_1_3_2_1_54_1","first-page":"3","volume-title":"Eighth","author":"Top\u00e7uo\u011flu H.","year":"1999"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0022-0000(75)80008-0"},{"key":"e_1_3_2_1_56_1","volume-title":"2014 IEEE International Symposium on. IEEE","author":"Wang J.","year":"2014"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750393"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2016.57"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/71.80160"},{"key":"e_1_3_2_1_60_1","volume-title":"GameSoundCon, 2016","author":"Wakeland C.","year":"2016"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2010.216"}],"event":{"name":"PPoPP '18: 23nd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","location":"Vienna Austria","acronym":"PPoPP '18","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages","SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the 11th Workshop on General Purpose GPUs"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3180270.3180271","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3180270.3180271","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T01:39:30Z","timestamp":1750210770000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3180270.3180271"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,2,24]]},"references-count":61,"alternative-id":["10.1145\/3180270.3180271","10.1145\/3180270"],"URL":"https:\/\/doi.org\/10.1145\/3180270.3180271","relation":{},"subject":[],"published":{"date-parts":[[2018,2,24]]},"assertion":[{"value":"2018-02-24","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}