{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,4]],"date-time":"2025-05-04T09:40:05Z","timestamp":1746351605464,"version":"3.40.4"},"reference-count":24,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2014,8,22]],"date-time":"2014-08-22T00:00:00Z","timestamp":1408665600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Parallel Prog"],"published-print":{"date-parts":[[2016,2]]},"DOI":"10.1007\/s10766-014-0318-5","type":"journal-article","created":{"date-parts":[[2014,8,21]],"date-time":"2014-08-21T13:29:34Z","timestamp":1408627774000},"page":"109-129","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["A Credit-Based Load-Balance-Aware CTA Scheduling Optimization Scheme in GPGPU"],"prefix":"10.1007","volume":"44","author":[{"given":"Yulong","family":"Yu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xubin","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"He","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xin","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,8,22]]},"reference":[{"doi-asserted-by":"crossref","unstructured":"Narasiman, V., Shebanow, M., Lee, C. et al.: Improving GPU performance via large warps and two-level warp scheduling. In: International Symposium on Microarchitecture, pp. 308\u2013317 (2011)","key":"318_CR1","DOI":"10.1145\/2155620.2155656"},{"doi-asserted-by":"crossref","unstructured":"Jog, A., Kayiran, O., Nachiappan, N. et al.: OWL: cooperative thread array aware scheduling techniques for improving GPGPU performance. In: International Conference on Architectural Support for Programming Languages and Operating Systems, pp. 395\u2013406 (2013)","key":"318_CR2","DOI":"10.1145\/2451116.2451158"},{"doi-asserted-by":"crossref","unstructured":"Kayiran, O., Jog, A., Kandermir, M. et al.: Neither more nor less: optimizing thread-level parallelism for GPGPUs. In: International Conference on Parallel Architectures and Compilation Techniques, pp. 157\u2013166 (2013)","key":"318_CR3","DOI":"10.1109\/PACT.2013.6618806"},{"unstructured":"NVIDIA: CUDA C Programming Guide (2012) http:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/index.html","key":"318_CR4"},{"unstructured":"Khronos Group: The open standard for parallel programming of heterogeneous systems (2013) http:\/\/www.khronos.org\/opencl\/","key":"318_CR5"},{"unstructured":"NVIDIA: NVIDIA Visual Profiler (2014) https:\/\/developer.nvidia.com\/nvidia-visual-profiler","key":"318_CR6"},{"unstructured":"NVIDIA: CUDA C\/C++ SDK code samples (2011) http:\/\/www.nvidia.com\/cuda-cc-sdk-code-samples","key":"318_CR7"},{"doi-asserted-by":"crossref","unstructured":"Bakhoda, A., Yuan, G., Fung, W. et al.: Analyzing CUDA workloads using a detailed GPU simulator. In: International Symposium on Performance Analysis of Systems and Software, pp. 163\u2013174 (2009)","key":"318_CR8","DOI":"10.1109\/ISPASS.2009.4919648"},{"unstructured":"NVIDIA: Tesla C2050 \/ C2070 GPU computing processor (2010). http:\/\/www.nvidia.com\/docs\/IO\/43395\/NV_DS_Tesla_C2050_C2070_jul10_lores.pdf","key":"318_CR9"},{"doi-asserted-by":"crossref","unstructured":"Che, S., Boyer, M., Meng, J. et al.: Rodinia: a benchmark suite for heterogeneous computing. In: International Symposium on Workload Characterization, pp. 44\u201354 (2009)","key":"318_CR10","DOI":"10.1109\/IISWC.2009.5306797"},{"unstructured":"Stratton, J.A., Rodrigues, C., Sung, I.J. et al.: Parboil: a revised benchmark suite for scientific and commercial throughput computing. Tech. Rep. IMPACT-12-01 University of Illinois at Urbana-Champaign (2012)","key":"318_CR11"},{"doi-asserted-by":"crossref","unstructured":"Lee, M., Song, S., Moon, J. et al.: Improving GPGPU resource utilization through alternative thread block scheduling. In: International Symposium on High Performance Computer Architecture, pp. 263\u2013273 (2014)","key":"318_CR12","DOI":"10.1109\/HPCA.2014.6835937"},{"doi-asserted-by":"crossref","unstructured":"Adriaens, J., Compton, K., Kim, N. et al.: The case for GPGPU spatial multitasking. In: International Symposium on High Performance Computer Architecture, pp. 1\u201312 (2012)","key":"318_CR13","DOI":"10.1109\/HPCA.2012.6168946"},{"doi-asserted-by":"crossref","unstructured":"Jog, A., Kayiran, O., Mishra, A. et al.: Orchestrated scheduling and prefetching for GPGPUs. In: International Symposium on Computer Architecture, pp. 332\u2013343 (2013)","key":"318_CR14","DOI":"10.1145\/2485922.2485951"},{"doi-asserted-by":"crossref","unstructured":"Gebhart, M., Johnson, D.R., Tarjan, D. et al.: Energy-efficient mechanisms for managing thread context in throughput processors. In: International Symposium on Computer Architecture, pp. 235\u2013246 (2011)","key":"318_CR15","DOI":"10.1145\/2000064.2000093"},{"doi-asserted-by":"crossref","unstructured":"Rogers, T., O\u2019Connor, M., Aamodt, T. et al.: Cache-conscious wavefront scheduling. In: International Symposium on Microarchitecture, pp. 72\u201383 (2012)","key":"318_CR16","DOI":"10.1109\/MICRO.2012.16"},{"doi-asserted-by":"crossref","unstructured":"Meng, J., Tarjan, D., Skadron, K.: Dynamic warp subdivision for integrated branch and memory divergence tolerance. In: International Symposium on Computer Architecture, pp. 235\u2013246 (2010)","key":"318_CR17","DOI":"10.1145\/1815961.1815992"},{"doi-asserted-by":"crossref","unstructured":"Fung, W.W.L., Sham, I., Yuan, G. et al.: Dynamic warp formation and scheduling for efficient GPU control flow. In: International Symposium on Microarchitecture, pp. 407\u2013420 (2007)","key":"318_CR18","DOI":"10.1109\/MICRO.2007.30"},{"doi-asserted-by":"crossref","unstructured":"Fung, W., Aamodt, T.: Thread block compaction for efficient SIMT control flow. In: International Symposium on High Performance Computer Architecture, pp. 25\u201336 (2011)","key":"318_CR19","DOI":"10.1109\/HPCA.2011.5749714"},{"doi-asserted-by":"crossref","unstructured":"Brunie, N., Collange, S., Diamos, G.: Simultaneous branch and warp interweaving for sustained GPU performance. In: International Symposium on Computer Architecture, pp. 49\u201360 (2012)","key":"318_CR20","DOI":"10.1109\/ISCA.2012.6237005"},{"doi-asserted-by":"crossref","unstructured":"Jia, W., Shaw, K.A., Martonosi, M.: MRPB: memory request prioritization for massively parallel processors. In: International Symposium on High Performance Computer Architecture, pp. 274\u2013285 (2014)","key":"318_CR21","DOI":"10.1109\/HPCA.2014.6835938"},{"doi-asserted-by":"crossref","unstructured":"Jog, A., Bolotin, E., Guz, Z. et al.: Application-aware memory system for fair and efficient execution of concurrent GPGPU applications. In: Workshop on General Purpose Processing Using GPUs, pp. 1\u20138 (2014)","key":"318_CR22","DOI":"10.1145\/2588768.2576780"},{"unstructured":"Lakshminarayana, N.B., Kim, H.: Effect of instruction fetch and memory scheduling on GPU performance. In: Workshop on Language, Compiler, and Architecture Support for GPGPU, pp. 1\u201310 (2010)","key":"318_CR23"},{"doi-asserted-by":"crossref","unstructured":"Chen, L., Villa, O., Krishnamoorthy, S. et al.: Dynamic load balancing on single- and multi-GPU systems. In: IEEE International Symposium on Parallel and Distributed Processing, pp. 1\u201312 (2010)","key":"318_CR24","DOI":"10.1109\/IPDPS.2010.5470413"}],"container-title":["International Journal of Parallel Programming"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-014-0318-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10766-014-0318-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-014-0318-5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,4]],"date-time":"2025-05-04T09:06:16Z","timestamp":1746349576000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10766-014-0318-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,8,22]]},"references-count":24,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2016,2]]}},"alternative-id":["318"],"URL":"https:\/\/doi.org\/10.1007\/s10766-014-0318-5","relation":{},"ISSN":["0885-7458","1573-7640"],"issn-type":[{"type":"print","value":"0885-7458"},{"type":"electronic","value":"1573-7640"}],"subject":[],"published":{"date-parts":[[2014,8,22]]}}}