{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T07:21:24Z","timestamp":1777965684455,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2016,11,15]],"date-time":"2016-11-15T00:00:00Z","timestamp":1479168000000},"content-version":"vor","delay-in-days":366,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["SHF-1217917"],"award-info":[{"award-number":["SHF-1217917"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000185","name":"Defense Advanced Research Projects Agency","doi-asserted-by":"publisher","award":["Power Efficiency Revolution for Embedded Computing Technologies (PERFECT) program"],"award-info":[{"award-number":["Power Efficiency Revolution for Embedded Computing Technologies (PERFECT) program"]}],"id":[{"id":"10.13039\/100000185","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2015,11,15]]},"DOI":"10.1145\/2807591.2807598","type":"proceedings-article","created":{"date-parts":[[2015,10,27]],"date-time":"2015-10-27T09:07:31Z","timestamp":1445936851000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":13,"title":["ELF"],"prefix":"10.1145","author":[{"given":"Jason Jong Kyu","family":"Park","sequence":"first","affiliation":[{"name":"University of Michigan, Ann Arbor, MI"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongjun","family":"Park","sequence":"additional","affiliation":[{"name":"Hongik University, Seoul, Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Scott","family":"Mahlke","sequence":"additional","affiliation":[{"name":"University of Michigan, Ann Arbor, MI"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2015,11,15]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2009.4919648"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.5555\/998680.1006708"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/263580.263597"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/2000064.2000093"},{"key":"e_1_3_2_1_6_1","volume-title":"MLP yes! ILP no!","author":"Glew A.","year":"1998","unstructured":"A. Glew. MLP yes! ILP no!, 1998. In ASPLOS Wild and Crazy Idea Session'98."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/2628071.2628072"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2014.6835938"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/2451116.2451158"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.5555\/2523721.2523745"},{"key":"e_1_3_2_1_11_1","volume-title":"OpenCL - the open standard for parallel programming of heterogeneous systems","author":"KHRONOS Group","year":"2010","unstructured":"KHRONOS Group. OpenCL - the open standard for parallel programming of heterogeneous systems, 2010."},{"key":"e_1_3_2_1_12_1","first-page":"1","volume-title":"Workshop on Language, Compiler, and Architecture Support for GPGPU","author":"Lakshminarayana N. B.","year":"2010","unstructured":"N. B. Lakshminarayana and H. Kim. Effect of instruction fetch and memory scheduling on GPU performance. In Workshop on Language, Compiler, and Architecture Support for GPGPU, pages 1--10, 2010."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2010.44"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2014.6835937"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2012.37"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.5555\/822080.822823"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/2155620.2155656"},{"key":"e_1_3_2_1_18_1","unstructured":"NVIDIA. GPU Computing SDK. http:\/\/developer.nvidia.com\/gpu-computing-sdk."},{"key":"e_1_3_2_1_19_1","volume-title":"Fermi: Nvidia\u015b next generation CUDA compute architecture","author":"NVIDIA.","year":"2009","unstructured":"NVIDIA. Fermi: Nvidia\u015b next generation CUDA compute architecture, 2009. http:\/\/www.nvidia.com\/content\/PDF\/fermi white papers\/NVIDIA_Fermi_Compute_Architecture_Whitepaper.pdf."},{"key":"e_1_3_2_1_20_1","volume-title":"May","author":"NVIDIA.","year":"2011","unstructured":"NVIDIA. CUDA C Programming Guide, May 2011."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.5555\/320080.320103"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2006.5"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2012.16"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2540708.2540718"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/2254064.2254067"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.5555\/2337159.2337210"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.5555\/2523721.2523735"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2015.7056031"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.5555\/147877.148093"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/279358.279408"},{"key":"e_1_3_2_1_31_1","unstructured":"J. A. Stratton C. Rodrigues I.-J. Sung N. Obeid L.-W. Chang N. Anssari G. D. Liu andW. mei Hwu. Parboil: A revised benchmark suite for scientific and commercial throughput computing. Technical Report IMPACT-12-01 University of Illinois at Urbana-Champaign Mar. 2012."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2005.117"}],"event":{"name":"SC15: The International Conference for High Performance Computing, Networking, Storage and Analysis","location":"Austin Texas","acronym":"SC15","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture","IEEE-CS Computer Society"]},"container-title":["Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2807591.2807598","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2807591.2807598","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2807591.2807598","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T09:43:50Z","timestamp":1763459030000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2807591.2807598"}},"subtitle":["maximizing memory-level parallelism for GPUs with coordinated warp and fetch scheduling"],"short-title":[],"issued":{"date-parts":[[2015,11,15]]},"references-count":32,"alternative-id":["10.1145\/2807591.2807598","10.1145\/2807591"],"URL":"https:\/\/doi.org\/10.1145\/2807591.2807598","relation":{},"subject":[],"published":{"date-parts":[[2015,11,15]]},"assertion":[{"value":"2015-11-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}