{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,4]],"date-time":"2025-11-04T16:08:55Z","timestamp":1762272535210},"reference-count":23,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2020,3,25]],"date-time":"2020-03-25T00:00:00Z","timestamp":1585094400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,3,25]],"date-time":"2020-03-25T00:00:00Z","timestamp":1585094400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Parallel Prog"],"published-print":{"date-parts":[[2020,12]]},"DOI":"10.1007\/s10766-020-00658-y","type":"journal-article","created":{"date-parts":[[2020,3,25]],"date-time":"2020-03-25T21:02:34Z","timestamp":1585170154000},"page":"1008-1031","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Memory-Optimized Wavefront Parallelism on GPUs"],"prefix":"10.1007","volume":"48","author":[{"given":"Yuanzhe","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Loren","family":"Schwiebert","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,3,25]]},"reference":[{"key":"658_CR1","doi-asserted-by":"crossref","unstructured":"Balhaf, K., Alsmirat, M.A., Al-Ayyoub, M., Jararweh, Y., Shehab, M.A.: Accelerating levenshtein and damerau edit distance algorithms using gpu with unified memory. In: 2017 8th International Conference on Information and Communication Systems (ICICS), pp. 7\u201311. IEEE (2017)","DOI":"10.1109\/IACS.2017.7921937"},{"key":"658_CR2","doi-asserted-by":"publisher","first-page":"175","DOI":"10.1016\/j.is.2016.06.001","volume":"64","author":"D Bedn\u00e1rek","year":"2017","unstructured":"Bedn\u00e1rek, D., Brabec, M., Kruli\u0161, M.: Improving matrix-based dynamic programming on massively parallel accelerators. Inf. Syst. 64, 175\u2013193 (2017)","journal-title":"Inf. Syst."},{"key":"658_CR3","doi-asserted-by":"crossref","unstructured":"Belviranli, M.E., Deng, P., Bhuyan, L.N., Gupta, R., Zhu, Q.: Peerwave: Exploiting wavefront parallelism on GPUS with peer-SM synchronization. In: Proceedings of the 29th ACM on International Conference on Supercomputing, pp. 25\u201335. ACM (2015)","DOI":"10.1145\/2751205.2751243"},{"key":"658_CR4","doi-asserted-by":"crossref","unstructured":"Crow, F.C.: Summed-area tables for texture mapping. In: ACM SIGGRAPH Computer Graphics, vol.\u00a018, pp. 207\u2013212. ACM (1984)","DOI":"10.1145\/964965.808600"},{"issue":"8","key":"658_CR5","doi-asserted-by":"publisher","first-page":"673","DOI":"10.1002\/spe.976","volume":"40","author":"S Deorowicz","year":"2010","unstructured":"Deorowicz, S.: Solving longest common subsequence and related problems on graphical processing units. Softw. Pract. Exp. 40(8), 673\u2013700 (2010)","journal-title":"Softw. Pract. Exp."},{"issue":"6\u20137","key":"658_CR6","doi-asserted-by":"publisher","first-page":"310","DOI":"10.1016\/j.parco.2012.03.004","volume":"38","author":"P Di","year":"2012","unstructured":"Di, P., Wu, H., Xue, J., Wang, F., Yang, C.: Parallelizing sor for gpgpus using alternate loop tiling. Parallel Comput. 38(6\u20137), 310\u2013328 (2012)","journal-title":"Parallel Comput."},{"key":"658_CR7","doi-asserted-by":"crossref","unstructured":"Di, P., Xue, J.: Model-driven tile size selection for doacross loops on GPUS. In: European Conference on Parallel Processing, pp. 401\u2013412. Springer (2011)","DOI":"10.1007\/978-3-642-23397-5_40"},{"key":"658_CR8","doi-asserted-by":"crossref","unstructured":"Di, P., Ye, D., Su, Y., Sui, Y., Xue, J.: Automatic parallelization of tiled loop nests with enhanced fine-grained parallelism on GPUS. In: 2012 41st International Conference on Parallel Processing, pp. 350\u2013359. IEEE (2012)","DOI":"10.1109\/ICPP.2012.19"},{"key":"658_CR9","volume-title":"Introduction to Parallel Computing","author":"A Grama","year":"2003","unstructured":"Grama, A., Kumar, V., Gupta, A., Karypis, G.: Introduction to Parallel Computing. Pearson Education, London (2003)"},{"key":"658_CR10","doi-asserted-by":"crossref","unstructured":"Hou, K., Wang, H., Feng, W.c., Vetter, J.S., Lee, S.: Highly efficient compensation-based parallelism for wavefront loops on GPUS. In: 2018 IEEE International Parallel and Distributed Processing Symposium (IPDPS), pp. 276\u2013285. IEEE (2018)","DOI":"10.1109\/IPDPS.2018.00037"},{"issue":"11","key":"658_CR11","doi-asserted-by":"publisher","first-page":"4247","DOI":"10.1016\/j.jcp.2010.02.009","volume":"229","author":"A Khajeh-Saeed","year":"2010","unstructured":"Khajeh-Saeed, A., Poole, S., Perot, J.B.: Acceleration of the smith\u2013waterman algorithm using single and multiple graphics processors. J. Comput. Phys. 229(11), 4247\u20134258 (2010)","journal-title":"J. Comput. Phys."},{"key":"658_CR12","doi-asserted-by":"crossref","unstructured":"Li, X., Liang, Y., Yan, S., Jia, L., Li, Y.: A coordinated tiling and batching framework for efficient GEMM on GPUS. In: Proceedings of the 24th Symposium on Principles and Practice of Parallel Programming, pp. 229\u2013241. ACM (2019)","DOI":"10.1145\/3293883.3295734"},{"key":"658_CR13","doi-asserted-by":"crossref","unstructured":"Li, Y., Ghalami, L., Schwiebert, L., Grosu, D.: A GPU parallel approximation algorithm for scheduling parallel identical machines to minimize makespan. In: 2018 IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW), pp. 619\u2013628. IEEE (2018)","DOI":"10.1109\/IPDPSW.2018.00102"},{"key":"658_CR14","doi-asserted-by":"crossref","unstructured":"Li, Z., Goyal, A., Kimm, H.: Parallel longest common sequence algorithm on multicore systems using openacc, openmp and openmpi. In:2017 IEEE 11th International Symposium on Embedded Multicore\/Many-core Systems-on-Chip (MCSoC), pp. 158\u2013165. IEEE (2017)","DOI":"10.1109\/MCSoC.2017.13"},{"issue":"1","key":"658_CR15","doi-asserted-by":"publisher","first-page":"117","DOI":"10.1186\/1471-2105-14-117","volume":"14","author":"Y Liu","year":"2013","unstructured":"Liu, Y., Wirawan, A., Schmidt, B.: Cudasw++ 3.0: accelerating smith\u2013waterman protein database search by coupling CPU and GPU SIMD instructions. BMC Bioinform. 14(1), 117 (2013)","journal-title":"BMC Bioinform."},{"issue":"4","key":"658_CR16","doi-asserted-by":"publisher","first-page":"C439","DOI":"10.1137\/140991133","volume":"37","author":"T Malas","year":"2015","unstructured":"Malas, T., Hager, G., Ltaief, H., Stengel, H., Wellein, G., Keyes, D.: Multicore-optimized wavefront diamond blocking for optimizing stencil updates. SIAM J. Sci. Comput. 37(4), C439\u2013C464 (2015)","journal-title":"SIAM J. Sci. Comput."},{"key":"658_CR17","unstructured":"Nvidia Corporation: CUDA Programming Guide (2018)"},{"key":"658_CR18","doi-asserted-by":"crossref","unstructured":"Rawat, P.S., Hong, C., Ravishankar, M., Grover, V., Pouchet, L.N., Rountev, A., Sadayappan, P.: Resource conscious reuse-driven tiling for GPUS. In: Proceedings of the 2016 International Conference on Parallel Architectures and Compilation, pp. 99\u2013111. ACM (2016)","DOI":"10.1145\/2967938.2967967"},{"key":"658_CR19","doi-asserted-by":"crossref","unstructured":"Remmelg, T., Lutz, T., Steuwer, M., Dubach, C.: Performance portable GPU code generation for matrix multiplication. In: Proceedings of the 9th Annual Workshop on General Purpose Processing using Graphics Processing Unit, pp. 22\u201331. ACM (2016)","DOI":"10.1145\/2884045.2884046"},{"issue":"4","key":"658_CR20","doi-asserted-by":"publisher","first-page":"482","DOI":"10.1016\/0196-8858(81)90046-4","volume":"2","author":"TF Smith","year":"1981","unstructured":"Smith, T.F., Waterman, M.S.: Comparison of biosequences. Adv. Appl. Math. 2(4), 482\u2013489 (1981)","journal-title":"Adv. Appl. Math."},{"key":"658_CR21","doi-asserted-by":"crossref","unstructured":"Striemer, G.M., Akoglu, A.: Sequence alignment with GPU: performance and design challenges (2009)","DOI":"10.1109\/IPDPS.2009.5161066"},{"key":"658_CR22","doi-asserted-by":"crossref","unstructured":"Tang, Y., You, R., Kan, H., Tithi, J.J., Ganapathi, P., Chowdhury, R.A.: Cache-oblivious wavefront: Improving parallelism of recursive dynamic programming algorithms without losing cache-efficiency. In: ACM SIGPLAN Notices, vol.\u00a050, pp. 205\u2013214. ACM (2015)","DOI":"10.1145\/2858788.2688514"},{"issue":"4","key":"658_CR23","doi-asserted-by":"publisher","first-page":"452","DOI":"10.1109\/71.97902","volume":"2","author":"ME Wolf","year":"1991","unstructured":"Wolf, M.E., Lam, M.S.: A loop transformation theory and an algorithm to maximize parallelism. IEEE Trans. Parallel Distrib. Syst. 2(4), 452\u2013471 (1991)","journal-title":"IEEE Trans. Parallel Distrib. Syst."}],"container-title":["International Journal of Parallel Programming"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-020-00658-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10766-020-00658-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-020-00658-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,3,25]],"date-time":"2021-03-25T00:52:45Z","timestamp":1616633565000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10766-020-00658-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,3,25]]},"references-count":23,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2020,12]]}},"alternative-id":["658"],"URL":"https:\/\/doi.org\/10.1007\/s10766-020-00658-y","relation":{},"ISSN":["0885-7458","1573-7640"],"issn-type":[{"value":"0885-7458","type":"print"},{"value":"1573-7640","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,3,25]]},"assertion":[{"value":"7 June 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 March 2020","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 March 2020","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}